@theglitchking/babel-fish 2.0.2 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/project-map/generate.py +206 -21
- package/.claude/project-map/grader.py +77 -1
- package/.claude/project-map/mine-sessions.py +141 -23
- package/.claude/project-map/test_generate.py +203 -0
- package/.claude/project-map/test_grader.py +164 -0
- package/.claude/project-map/test_mine_sessions.py +206 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/.claude-plugin/plugin.json +1 -1
- package/CHANGELOG.md +209 -0
- package/README.md +9 -2
- package/checksums.json +2 -2
- package/hooks/session-start.js +34 -1
- package/package.json +13 -4
- package/.claude/project-map/PROJECT_MAP.md +0 -61
- package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
- package/.claude/project-map/checksums.json +0 -6
- package/.claude/project-map/learned-vocabulary.json +0 -1
- package/.claude/project-map/reports/install-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-score.json +0 -7
- package/.claude/project-map/sections/01-vocabulary.md +0 -8
- package/.claude/project-map/sections/02-service-topology.md +0 -6
- package/.claude/project-map/sections/03-environment.md +0 -6
- package/.claude/project-map/sections/04-api-routes.md +0 -6
- package/.claude/project-map/sections/05-data-models.md +0 -4
- package/.claude/project-map/sections/06-schemas.md +0 -4
- package/.claude/project-map/sections/07-services.md +0 -6
- package/.claude/project-map/sections/08-background-jobs.md +0 -5
- package/.claude/project-map/sections/09-frontend-features.md +0 -4
- package/.claude/project-map/sections/10-tools-commands.md +0 -8
- package/.claude/project-map/sections/11-migrations.md +0 -4
- package/.claude/project-map/sections/12-import-chains.md +0 -7
- package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
- package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
- package/.claude/project-map/sections/15-auth-config.md +0 -6
- package/.claude/project-map/sections/16-infra-profile.md +0 -13
- package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
- package/.claude/project-map/sections/18-dead-code.md +0 -9
- package/.claude/project-map/sections/19-doc-pointers.md +0 -5
- package/.claude/project-map/stack.json +0 -12
- package/.claude/rules/operational-runbook.md +0 -40
- package/.claude/rules/project-vocabulary.md +0 -25
- package/.claude/settings.json +0 -6
- package/.claude/settings.local.json +0 -6
- package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Regression tests for mine-sessions.py — run: python3 .claude/project-map/test_mine_sessions.py
|
|
3
|
+
|
|
4
|
+
Same shape as test_generate.py: stdlib assert + __main__, no framework.
|
|
5
|
+
|
|
6
|
+
Every test here corresponds to a defect that shipped and produced NO error —
|
|
7
|
+
the miner exited 0 and reported "Extracted 0 alias(es)" for its entire
|
|
8
|
+
existence. Silent-zero is the failure mode these guard against.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import importlib.util
|
|
13
|
+
import json
|
|
14
|
+
import shutil
|
|
15
|
+
import sys
|
|
16
|
+
import tempfile
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
HERE = Path(__file__).parent
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load_miner(project_root: Path, cursor_dir: Path):
|
|
24
|
+
spec = importlib.util.spec_from_file_location("miner_under_test", HERE / "mine-sessions.py")
|
|
25
|
+
mod = importlib.util.module_from_spec(spec)
|
|
26
|
+
argv, sys.argv = sys.argv, ["mine-sessions.py"]
|
|
27
|
+
try:
|
|
28
|
+
spec.loader.exec_module(mod)
|
|
29
|
+
finally:
|
|
30
|
+
sys.argv = argv
|
|
31
|
+
mod.PROJECT_ROOT = project_root
|
|
32
|
+
mod.MINE_CURSOR = cursor_dir / ".mine-cursor.json"
|
|
33
|
+
return mod
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def write_transcript(path: Path, entries: list[dict]) -> None:
|
|
37
|
+
path.write_text("\n".join(json.dumps(e) for e in entries) + "\n")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def user_msg(text: str, **extra) -> dict:
|
|
41
|
+
"""A genuine user turn, in the real nested shape Claude Code writes."""
|
|
42
|
+
return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
43
|
+
"message": {"role": "user", "content": [{"type": "text", "text": text}]}, **extra}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def tool_msg(name: str, inp: dict) -> dict:
|
|
47
|
+
return {"type": "assistant", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
48
|
+
"message": {"role": "assistant",
|
|
49
|
+
"content": [{"type": "tool_use", "name": name, "input": inp}]}}
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def tool_result() -> dict:
|
|
53
|
+
"""Tool results come back with role=user — the collision that broke pairing."""
|
|
54
|
+
return {"type": "user", "timestamp": datetime.now(timezone.utc).isoformat(),
|
|
55
|
+
"message": {"role": "user",
|
|
56
|
+
"content": [{"type": "tool_result", "content": "ok"}]}}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
# ── Defect 1: JSONL nesting ──────────────────────────────────────────────────
|
|
60
|
+
|
|
61
|
+
def test_reads_nested_message_content(m, root):
|
|
62
|
+
msg = user_msg("open the settings page")
|
|
63
|
+
assert m.msg_role(msg) == "user", "role must be read from message.role"
|
|
64
|
+
content = m.msg_content(msg)
|
|
65
|
+
assert isinstance(content, list) and content[0]["text"] == "open the settings page"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_still_reads_flat_content(m, root):
|
|
69
|
+
"""Older/third-party transcripts must not regress."""
|
|
70
|
+
flat = {"role": "user", "content": [{"type": "text", "text": "hello there"}]}
|
|
71
|
+
assert m.msg_role(flat) == "user"
|
|
72
|
+
assert m.msg_content(flat)[0]["text"] == "hello there"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_extracts_tool_paths_from_nested(m, root):
|
|
76
|
+
paths = m.extract_file_paths_from_tool_calls(
|
|
77
|
+
[tool_msg("Read", {"file_path": str(root / "src/app.py")})])
|
|
78
|
+
assert paths == ["src/app.py"], paths
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ── Defect 2: regex quantifiers ──────────────────────────────────────────────
|
|
82
|
+
|
|
83
|
+
def test_the_x_page_pattern_matches(m, root):
|
|
84
|
+
"""'{2,40?}' compiled fine and matched nothing — no error, just silence."""
|
|
85
|
+
got = m.extract_user_phrases("please look at the settings page")
|
|
86
|
+
assert "settings" in got, got
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def test_x_feature_pattern_matches(m, root):
|
|
90
|
+
got = m.extract_user_phrases("the billing workflow is broken")
|
|
91
|
+
assert any("billing" in g for g in got), got
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ── Defect 3: tool_result / user-turn collision ──────────────────────────────
|
|
95
|
+
|
|
96
|
+
def test_tool_result_is_not_a_user_turn(m, root):
|
|
97
|
+
assert m.is_user_turn(user_msg("the deals page")) is True
|
|
98
|
+
assert m.is_user_turn(tool_result()) is False, (
|
|
99
|
+
"tool results carry role=user; treating them as user turns closed the "
|
|
100
|
+
"pairing window on the assistant's own output"
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def test_meta_and_sidechain_are_not_user_turns(m, root):
|
|
105
|
+
assert m.is_user_turn(user_msg("skill body text", isMeta=True)) is False
|
|
106
|
+
assert m.is_user_turn(user_msg("subagent text", isSidechain=True)) is False
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ── Defect 4: Bash-mediated file access ──────────────────────────────────────
|
|
110
|
+
|
|
111
|
+
def test_bash_paths_extracted_when_file_exists(m, root):
|
|
112
|
+
(root / "src").mkdir(parents=True, exist_ok=True)
|
|
113
|
+
(root / "src/app.py").write_text("# app\n")
|
|
114
|
+
got = m.extract_paths_from_bash("sed -n '1,40p' src/app.py", root)
|
|
115
|
+
assert got == ["src/app.py"], got
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def test_bash_paths_ignore_nonexistent(m, root):
|
|
119
|
+
got = m.extract_paths_from_bash("cat totally/made/up.py && ls -la", root)
|
|
120
|
+
assert got == [], f"only real files may become aliases, got {got}"
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
# ── Defect 6: session discovery ──────────────────────────────────────────────
|
|
124
|
+
|
|
125
|
+
def test_slug_keeps_leading_separator(m, root, monkey_home):
|
|
126
|
+
""".lstrip('-') meant the exact match never hit, so every lookup fell
|
|
127
|
+
through to a fuzzy substring match that ALSO ran additively. A decoy
|
|
128
|
+
sharing the project name proves the exact path is used and that another
|
|
129
|
+
project's transcripts are not swept in — "kentro" matches four real
|
|
130
|
+
directories on this machine."""
|
|
131
|
+
projects = monkey_home / ".claude" / "projects"
|
|
132
|
+
slug = str(root).replace("/", "-")
|
|
133
|
+
(projects / slug).mkdir(parents=True)
|
|
134
|
+
write_transcript(projects / slug / "s.jsonl", [user_msg("hi there")])
|
|
135
|
+
|
|
136
|
+
decoy = projects / (slug + "-other-project")
|
|
137
|
+
decoy.mkdir(parents=True)
|
|
138
|
+
write_transcript(decoy / "d.jsonl", [user_msg("decoy transcript")])
|
|
139
|
+
|
|
140
|
+
found = m.find_session_files(root)
|
|
141
|
+
assert len(found) == 1, f"expected only the exact match, got {found}"
|
|
142
|
+
assert found[0].parent.name == slug, found[0]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
# ── End to end ───────────────────────────────────────────────────────────────
|
|
146
|
+
|
|
147
|
+
def test_mines_alias_to_path(m, root, monkey_home):
|
|
148
|
+
"""The test that fails if any defect returns."""
|
|
149
|
+
(root / "src").mkdir(parents=True, exist_ok=True)
|
|
150
|
+
(root / "src/deals.py").write_text("# deals\n")
|
|
151
|
+
slug = str(root).replace("/", "-")
|
|
152
|
+
d = monkey_home / ".claude" / "projects" / slug
|
|
153
|
+
d.mkdir(parents=True)
|
|
154
|
+
|
|
155
|
+
entries = []
|
|
156
|
+
for _ in range(6): # clear MIN_SCORE = 5.0 at weight 1.0
|
|
157
|
+
entries.append(user_msg("update the deals page please"))
|
|
158
|
+
entries.append(tool_msg("Read", {"file_path": str(root / "src/deals.py")}))
|
|
159
|
+
entries.append(tool_result())
|
|
160
|
+
write_transcript(d / "s.jsonl", entries)
|
|
161
|
+
|
|
162
|
+
miner = m.SessionMiner(root)
|
|
163
|
+
miner.mine(m.find_session_files(root))
|
|
164
|
+
res = miner.results()
|
|
165
|
+
assert res, "mined nothing from a transcript containing 6 clear pairings"
|
|
166
|
+
assert "deals" in res, list(res)
|
|
167
|
+
assert "src/deals.py" in res["deals"]["targets"], res["deals"]
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def test_junk_phrases_filtered(m, root):
|
|
171
|
+
got = m.extract_user_phrases('he said "that, if not" and "total documents:"')
|
|
172
|
+
assert not any("," in g or ":" in g for g in got), got
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def main() -> int:
|
|
176
|
+
import inspect
|
|
177
|
+
tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
|
|
178
|
+
failed = []
|
|
179
|
+
tmp = Path(tempfile.mkdtemp(prefix="mine-test-"))
|
|
180
|
+
try:
|
|
181
|
+
for fn in tests:
|
|
182
|
+
root = tmp / fn.__name__ / "repo"
|
|
183
|
+
home = tmp / fn.__name__ / "home"
|
|
184
|
+
root.mkdir(parents=True); home.mkdir(parents=True)
|
|
185
|
+
m = load_miner(root, root)
|
|
186
|
+
params = inspect.signature(fn).parameters
|
|
187
|
+
kwargs = {}
|
|
188
|
+
needs_home = "monkey_home" in params
|
|
189
|
+
if needs_home:
|
|
190
|
+
# find_session_files() resolves ~/.claude/projects via Path.home()
|
|
191
|
+
m.Path.home = staticmethod(lambda: home)
|
|
192
|
+
kwargs["monkey_home"] = home
|
|
193
|
+
try:
|
|
194
|
+
fn(m, root, **kwargs)
|
|
195
|
+
print(f" ok {fn.__name__}")
|
|
196
|
+
except Exception as e:
|
|
197
|
+
failed.append(fn.__name__)
|
|
198
|
+
print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
|
|
199
|
+
finally:
|
|
200
|
+
shutil.rmtree(tmp, ignore_errors=True)
|
|
201
|
+
print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
|
|
202
|
+
return 1 if failed else 0
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
if __name__ == "__main__":
|
|
206
|
+
sys.exit(main())
|
|
@@ -6,13 +6,13 @@
|
|
|
6
6
|
},
|
|
7
7
|
"metadata": {
|
|
8
8
|
"description": "Official marketplace for babel-fish - Codebase introspection and vocabulary translation for AI coding assistants",
|
|
9
|
-
"version": "2.
|
|
9
|
+
"version": "2.3.0"
|
|
10
10
|
},
|
|
11
11
|
"plugins": [
|
|
12
12
|
{
|
|
13
13
|
"name": "babel-fish",
|
|
14
14
|
"description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
|
|
15
|
-
"version": "2.
|
|
15
|
+
"version": "2.3.0",
|
|
16
16
|
"author": {
|
|
17
17
|
"name": "TheGlitchKing"
|
|
18
18
|
},
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "babel-fish",
|
|
3
3
|
"description": "Auto-generates a project map, vocabulary translation layer, and developer skill for any codebase. Introspects routes, models, services, features, infrastructure, and session history to give Claude instant full-stack context. Self-updates via pre-commit hook.",
|
|
4
|
-
"version": "2.
|
|
4
|
+
"version": "2.3.0",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TheGlitchKing",
|
|
7
7
|
"email": "theglitchking@users.noreply.github.com"
|
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,215 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to this project will be documented in this file.
|
|
4
4
|
|
|
5
|
+
## [2.3.0] - 2026-09-04
|
|
6
|
+
|
|
7
|
+
### Fixed
|
|
8
|
+
|
|
9
|
+
- **The grader could not tell a useless map from a good one**
|
|
10
|
+
([#9](https://github.com/TheGlitchKing/babel-fish/issues/9)). All seven graded
|
|
11
|
+
categories measure *form*, and `generate.py` always emits well-formed output,
|
|
12
|
+
so the completely empty pre-2.1.0 map scored **97.0% PASS** — identical
|
|
13
|
+
category-for-category to the populated map. Verified against the real artifact
|
|
14
|
+
recovered from git, not a reconstruction.
|
|
15
|
+
|
|
16
|
+
Rather than reweighting, usefulness is now reported as **warnings that never
|
|
17
|
+
touch the score**, so no existing install flips from pass to fail:
|
|
18
|
+
|
|
19
|
+
- `generate.py` records *inputs* beside outputs (`### Source files scanned`).
|
|
20
|
+
"0 routes" cannot be judged alone; "0 routes from 47 Python files" can. The
|
|
21
|
+
counts already existed in `main()` and were being discarded.
|
|
22
|
+
- An empty vocabulary warns — that is what babel-fish is for, so zero entries
|
|
23
|
+
means the map gave you nothing. This fires on the real pre-2.1.0 map and is
|
|
24
|
+
the signal that would have surfaced #6 at install time.
|
|
25
|
+
- Ten or more source files scanned with nothing extracted warns separately,
|
|
26
|
+
catching a parser that does not fit the stack. Small repos stay quiet.
|
|
27
|
+
- A populated-sections count prints as an explicit diagnostic.
|
|
28
|
+
|
|
29
|
+
- **The greenfield branch in `grade_vocabulary_accuracy()` was unreachable.**
|
|
30
|
+
`build_vocabulary_section()` emitted a placeholder *table row* for an empty
|
|
31
|
+
vocabulary, which parsed as a valid entry whose blank location counts as
|
|
32
|
+
neutral — scoring 1/1 = 100% and stepping straight over the `if not rows`
|
|
33
|
+
branch written to award 85%. It now emits prose.
|
|
34
|
+
|
|
35
|
+
- **`or '_No' in content` matched too much.** Any italicised word beginning
|
|
36
|
+
"No" (`_Note`, `_Nothing`) anywhere in `12-import-chains.md` scored a
|
|
37
|
+
populated section as an acceptable empty one at a flat 80%.
|
|
38
|
+
|
|
39
|
+
### Fixed (packaging)
|
|
40
|
+
|
|
41
|
+
- **The npm tarball shipped local and generated files.** `files` listed
|
|
42
|
+
`.claude/` wholesale, and npm does **not** honour `.gitignore` for paths named
|
|
43
|
+
there — so every release carried this repo's own generated project map,
|
|
44
|
+
another plugin's local state (`.claude/.semantic-memory/`), five plugins'
|
|
45
|
+
update caches, ~150 kB of `__pycache__` bytecode, and
|
|
46
|
+
`.claude/settings.local.json`. `files` now lists only what
|
|
47
|
+
`.claude/install.sh` actually copies plus the test suites. Also anchored
|
|
48
|
+
`checksums.json` to `./checksums.json`: a bare filename in `files` globs at
|
|
49
|
+
any depth, so it was matching `.claude/project-map/checksums.json` too.
|
|
50
|
+
|
|
51
|
+
Tarball: 76 files / 140.9 kB → **33 files / 65.6 kB**. Verified by installing
|
|
52
|
+
from the packed tarball into a scratch project.
|
|
53
|
+
|
|
54
|
+
### Added
|
|
55
|
+
|
|
56
|
+
- 9 tests in `test_grader.py`, 7 of which fail against the previous code.
|
|
57
|
+
`npm test` now runs three suites (8 + 12 + 9 = 29).
|
|
58
|
+
- `architecture/grading-semantics.md` — what the score means, what it
|
|
59
|
+
deliberately omits, and the measurements behind that choice.
|
|
60
|
+
|
|
61
|
+
### Not done, deliberately
|
|
62
|
+
|
|
63
|
+
Issue #9 originally proposed scoring section completeness on populated content.
|
|
64
|
+
Measured and withdrawn: it fails the *correct* map too (85.2%), because 19
|
|
65
|
+
sections is aspirational — a plugin repo can never populate routes, models,
|
|
66
|
+
schemas or migrations, so `populated/19` tops out near 10/19 on a perfect map.
|
|
67
|
+
It would fail every legitimately sparse repo, a worse failure than the one it
|
|
68
|
+
fixes. The 90% threshold and the category weights are unchanged.
|
|
69
|
+
|
|
70
|
+
## [2.2.0] - 2026-09-04
|
|
71
|
+
|
|
72
|
+
### Fixed
|
|
73
|
+
|
|
74
|
+
- **Session vocabulary mining has never worked**
|
|
75
|
+
([#7](https://github.com/TheGlitchKing/babel-fish/issues/7)). The issue
|
|
76
|
+
reported that `mine-sessions.py` has no caller. It also had six defects, each
|
|
77
|
+
sufficient on its own to make it extract nothing — it exited 0 reporting
|
|
78
|
+
"Extracted 0 alias(es)" for its entire existence, which is why they survived.
|
|
79
|
+
|
|
80
|
+
1. **Wrong JSONL nesting.** Read `msg['content']`; Claude Code writes
|
|
81
|
+
`msg['message']['content']`. Measured on a real transcript: 0 vs 157
|
|
82
|
+
`tool_use` blocks, 0 vs 14 user messages. Both halves of the pairing were
|
|
83
|
+
empty.
|
|
84
|
+
2. **Invalid regex quantifiers.** `{2,40?}` and `{2,30?}` are malformed brace
|
|
85
|
+
expressions that Python silently treats as literals, so both patterns
|
|
86
|
+
compiled and matched nothing — including the one implementing this
|
|
87
|
+
feature's own README example, "the numbers page".
|
|
88
|
+
3. **Tool results collide with user turns.** Results arrive as `role: "user"`
|
|
89
|
+
(167 of 181 in one transcript), so the pairing window closed on the
|
|
90
|
+
assistant's own output.
|
|
91
|
+
4. **Bash file access was invisible.** A real session ran 148 Bash calls
|
|
92
|
+
against 2 Read and 2 Edit; only 4 of 157 tool calls qualified.
|
|
93
|
+
5. **Injected text was mined as user speech.** Skill and slash-command bodies
|
|
94
|
+
arrive in the user slot, and taught the miner aliases from the injected
|
|
95
|
+
documents themselves (`block_index_edits`, `refactor authentication
|
|
96
|
+
system` — the latter from a skill's worked example). Now skipped via
|
|
97
|
+
`isMeta` / `isSidechain`.
|
|
98
|
+
6. **Session discovery never matched exactly.** `.lstrip('-')` stripped the
|
|
99
|
+
leading separator that `~/.claude/projects/` slugs keep, so every lookup
|
|
100
|
+
fell through to a fuzzy substring match that also ran additively — and a
|
|
101
|
+
name like `kentro` matches four unrelated projects, whose aliases would be
|
|
102
|
+
attributed to this repo.
|
|
103
|
+
|
|
104
|
+
Verified against 220 MB of transcripts for a real product repo: 0 aliases
|
|
105
|
+
before, 170 after, reading like genuine domain vocabulary (`sign-up` →
|
|
106
|
+
`payments.py`, `pricing` → `subscription_gate.py`).
|
|
107
|
+
|
|
108
|
+
- **`README.md` claimed mining happened "automatically".** It did not — nothing
|
|
109
|
+
called the miner. Now true, and documented with its two real caveats.
|
|
110
|
+
|
|
111
|
+
### Added
|
|
112
|
+
|
|
113
|
+
- **Mining runs at session start.** `hooks/session-start.js` spawns the miner
|
|
114
|
+
detached with output discarded and nothing awaited; it cannot delay or fail a
|
|
115
|
+
session, and no-ops when Python or the script is absent.
|
|
116
|
+
- **Incremental cursor** (`.mine-cursor.json`). Not only a cost guard:
|
|
117
|
+
`merge_learned()` adds scores, so re-mining a counted transcript inflates it
|
|
118
|
+
without bound. `--all` forces a full re-mine.
|
|
119
|
+
- 12 tests in `test_mine_sessions.py`, one per defect plus an end-to-end mine.
|
|
120
|
+
All 12 fail against the previous miner. `npm test` runs both suites (20).
|
|
121
|
+
- Docs: `architecture/session-vocabulary-mining.md` (including the transcript
|
|
122
|
+
shape assumptions the miner depends on but does not control) and
|
|
123
|
+
`troubleshooting/learned-vocabulary-empty.md`.
|
|
124
|
+
|
|
125
|
+
### Known issues
|
|
126
|
+
|
|
127
|
+
- Aliases land one session late: SessionStart mines transcripts through the
|
|
128
|
+
previous session, since the current one isn't written yet.
|
|
129
|
+
- Phrase quality is heuristic. Filtering drops clause-like candidates, but a
|
|
130
|
+
quoted string in a user message can still become an alias.
|
|
131
|
+
|
|
132
|
+
## [2.1.1] - 2026-09-04
|
|
133
|
+
|
|
134
|
+
### Fixed
|
|
135
|
+
|
|
136
|
+
- **Section 19 listed generated hit-em-with-the-docs reports.**
|
|
137
|
+
`.documentation/reports/` holds timestamped audit output, so every `hewtd
|
|
138
|
+
maintain` wrote a new filename, which changed the doc path set, moved the
|
|
139
|
+
checksum and forced a full map regeneration — the exact churn the path-only
|
|
140
|
+
doc hash exists to prevent, reintroduced through a directory that was
|
|
141
|
+
gitignored but never excluded from the doc walk. `reports` joins `archive` in
|
|
142
|
+
`DOC_SKIP_DIRS`. Regression test added; it fails without the fix.
|
|
143
|
+
|
|
144
|
+
## [2.1.0] - 2026-09-04
|
|
145
|
+
|
|
146
|
+
### Fixed
|
|
147
|
+
|
|
148
|
+
- **Plugin and skill repositories no longer generate an empty project map**
|
|
149
|
+
([#6](https://github.com/TheGlitchKing/babel-fish/issues/6)). Run babel-fish
|
|
150
|
+
against a repo of markdown skills, slash commands and bash scripts and every
|
|
151
|
+
one of the 19 sections came back a "none detected" stub. Two causes, both
|
|
152
|
+
fixed:
|
|
153
|
+
|
|
154
|
+
- The checksum was blind to the files that define such a repo.
|
|
155
|
+
`collect_watched_files()` returned 11 files for babel-fish's own repository,
|
|
156
|
+
with no `.md` and no `.sh`, so editing a `SKILL.md` left the checksum
|
|
157
|
+
bit-identical and `is_unchanged()` exited before parsing. Skill and command
|
|
158
|
+
manifests are now matched by path glob (`skills/*/SKILL.md`,
|
|
159
|
+
`commands/*.md`), and `.sh` joins `WATCHED_EXTENSIONS`.
|
|
160
|
+
- Nothing read those manifests. `SkillParser` now feeds skill and command
|
|
161
|
+
frontmatter into section 01 (vocabulary) and section 10 (tools) — the two
|
|
162
|
+
sections they already fit. No new sections, no renumbering.
|
|
163
|
+
|
|
164
|
+
Measured on this repository: 0 vocabulary entries to 10, sections 2,589 bytes
|
|
165
|
+
to 4,389. On `hit-em-with-the-docs`, an unrelated plugin repo: 0 to 30.
|
|
166
|
+
|
|
167
|
+
- **Section 19 missed `.documentation/` trees and went stale silently.** Doc
|
|
168
|
+
directories were never watched, so adding a document did not move the
|
|
169
|
+
checksum and the pointer list rotted until an unrelated source file happened
|
|
170
|
+
to change. Doc paths are now hashed **without** mtime: adding, renaming or
|
|
171
|
+
deleting a document refreshes section 19, while editing one does not force a
|
|
172
|
+
full regeneration. `.documentation` joins the doc directories, and generated
|
|
173
|
+
navigation (`INDEX.md`, `REGISTRY.md`) plus `archive/` are excluded — without
|
|
174
|
+
that, a 15-domain tree contributes 32 nav files and crowds every real
|
|
175
|
+
document out of the 30-entry cap.
|
|
176
|
+
|
|
177
|
+
- **`checksums.json` was stale**, so the documented curl installer aborted with
|
|
178
|
+
`CHECKSUM MISMATCH` for everyone. `.claude/install.sh` was edited in `da8d2f7`
|
|
179
|
+
without regenerating the manifest.
|
|
180
|
+
|
|
181
|
+
### Added
|
|
182
|
+
|
|
183
|
+
- First tests in the repository: `npm test` runs an 8-test regression suite over
|
|
184
|
+
a fixture repo shaped like #6. Stdlib `assert`, no framework. Verified to fail
|
|
185
|
+
7/8 against the pre-fix generator rather than merely passing after it.
|
|
186
|
+
- `.documentation/` docs for the watch set, the skill parser contract, and an
|
|
187
|
+
empty/stale map troubleshooting guide; operational runbook gained the
|
|
188
|
+
corresponding gotchas.
|
|
189
|
+
|
|
190
|
+
### Known issues
|
|
191
|
+
|
|
192
|
+
- `grader.py` scores a completely empty map at 97.0% PASS, the same as a fully
|
|
193
|
+
populated one — "Vocabulary Accuracy" is 100% on zero entries because
|
|
194
|
+
0/0 = 100. It measures well-formedness, not usefulness, and must not be used
|
|
195
|
+
to confirm an extractor fix. Left unchanged here: adding a floor would fail
|
|
196
|
+
existing installs that currently pass.
|
|
197
|
+
- `01-vocabulary.md` is emitted as a markdown table, while
|
|
198
|
+
`.documentation/api/glossary-contract.md` specifies `- **key** → \`path\``
|
|
199
|
+
bullets. A consumer implemented strictly to that contract extracts zero
|
|
200
|
+
entries. Predates this release; which side moves is undecided.
|
|
201
|
+
|
|
202
|
+
## [2.0.3] - 2026-06-08
|
|
203
|
+
|
|
204
|
+
### Fixed
|
|
205
|
+
|
|
206
|
+
- Release-metadata consistency: the 2.0.2 npm payload shipped with
|
|
207
|
+
`.claude-plugin/plugin.json` still at version 2.0.0 (the manifest bump landed
|
|
208
|
+
after the 2.0.2 publish), so Claude Code read the installed plugin as 2.0.0
|
|
209
|
+
and `claude plugin update` never advanced. 2.0.3 bumps `package.json`,
|
|
210
|
+
`.claude-plugin/plugin.json`, and `.claude-plugin/marketplace.json` together
|
|
211
|
+
before publishing so the plugin version resolves correctly. No code change
|
|
212
|
+
from 2.0.2.
|
|
213
|
+
|
|
5
214
|
## [2.0.2] - 2026-06-08
|
|
6
215
|
|
|
7
216
|
### Fixed
|
package/README.md
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
[](https://github.com/TheGlitchKing/babel-fish)
|
|
10
10
|
|
|
11
11
|
> [!NOTE]
|
|
12
|
-
> **Pairs with [`semantic-memory`](https://github.com/TheGlitchKing/semantic-sidekick) (formerly `semantic-sidekick`).** When both are installed, semantic-memory consumes babel-fish's auto-generated `.babel-fish/` output as a `project-map` corpus AND extracts `01-vocabulary.md` into a structured `glossary.json` side-channel. That gives your AI a deterministic `translate("deals page") → "features/deal-pipeline/DealPipeline.tsx"` MCP verb instead of relying on semantic-search-luck. See [
|
|
12
|
+
> **Pairs with [`semantic-memory`](https://github.com/TheGlitchKing/semantic-sidekick) (formerly `semantic-sidekick`).** When both are installed, semantic-memory consumes babel-fish's auto-generated `.babel-fish/` output as a `project-map` corpus AND extracts `01-vocabulary.md` into a structured `glossary.json` side-channel. That gives your AI a deterministic `translate("deals page") → "features/deal-pipeline/DealPipeline.tsx"` MCP verb instead of relying on semantic-search-luck. See [`.documentation/api/glossary-contract.md`](./.documentation/api/glossary-contract.md) for the producer/consumer data contract and [`.documentation/quickstart/integration-with-semantic-memory.md`](./.documentation/quickstart/integration-with-semantic-memory.md) for the setup walkthrough. babel-fish standalone behavior is unchanged — semantic-memory is purely additive.
|
|
13
13
|
|
|
14
14
|
---
|
|
15
15
|
|
|
@@ -200,7 +200,14 @@ python .claude/project-map/grader.py
|
|
|
200
200
|
|
|
201
201
|
## Learned Vocabulary
|
|
202
202
|
|
|
203
|
-
Every AI session is mined for vocabulary. When you say "the numbers page" and the AI opens `DealAnalyzerV2.tsx`, that alias is recorded with a score (frequency × recency). Aliases with a score ≥ 5 appear in `17-learned-vocabulary.md
|
|
203
|
+
Every AI session is mined for vocabulary. When you say "the numbers page" and the AI opens `DealAnalyzerV2.tsx`, that alias is recorded with a score (frequency × recency). Aliases with a score ≥ 5 appear in `17-learned-vocabulary.md`.
|
|
204
|
+
|
|
205
|
+
Mining runs automatically at session start *(2.2.0+)*, detached and best-effort — it never delays or blocks a session. Only transcripts changed since the last run are read.
|
|
206
|
+
|
|
207
|
+
Two things to expect:
|
|
208
|
+
|
|
209
|
+
- **Aliases land one session late.** A session's transcript isn't written until it ends, so what you say today is mined at the *next* session start.
|
|
210
|
+
- **Operational sessions mine little.** The miner learns feature nouns — "the deals page", "the billing workflow". A session spent on refactoring or releases contains few of those and will correctly yield nothing. See [learned vocabulary is empty](./.documentation/troubleshooting/learned-vocabulary-empty.md).
|
|
204
211
|
|
|
205
212
|
Run the miner manually:
|
|
206
213
|
|
package/checksums.json
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
1
|
{
|
|
2
|
-
"install_sh": "
|
|
3
|
-
"note": "SHA256 of .claude/install.sh
|
|
2
|
+
"install_sh": "565ad35c264c8e683ade82f441ab5a9f8feb3921e0b4180623bb038b23452518",
|
|
3
|
+
"note": "SHA256 of .claude/install.sh \u2014 verified by the remote installer before execution"
|
|
4
4
|
}
|
package/hooks/session-start.js
CHANGED
|
@@ -1,11 +1,44 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
// babel-fish SessionStart hook. Runs the runtime's update check per
|
|
3
|
-
// policy (off / nudge / auto)
|
|
3
|
+
// policy (off / nudge / auto), then kicks off session vocabulary mining.
|
|
4
4
|
|
|
5
5
|
import { runSessionStart } from "@theglitchking/claude-plugin-runtime";
|
|
6
|
+
import { spawn, spawnSync } from "node:child_process";
|
|
7
|
+
import { existsSync } from "node:fs";
|
|
8
|
+
import { join } from "node:path";
|
|
9
|
+
|
|
10
|
+
// Mining is best-effort: it must never delay or fail session start, so it is
|
|
11
|
+
// detached with output discarded and nothing is awaited. It reads only
|
|
12
|
+
// transcripts changed since the last run (see mine-sessions.py's cursor).
|
|
13
|
+
//
|
|
14
|
+
// Timing: SessionStart sees transcripts through the PREVIOUS session — the
|
|
15
|
+
// current one isn't written yet — so an alias lands one session after it is
|
|
16
|
+
// first used. Accepted; a SessionEnd hook would be exact but is a second hook
|
|
17
|
+
// for one session of latency.
|
|
18
|
+
function mineVocabulary(cwd) {
|
|
19
|
+
try {
|
|
20
|
+
const script = join(cwd, ".claude", "project-map", "mine-sessions.py");
|
|
21
|
+
if (!existsSync(script)) return; // marketplace install that skipped the copy
|
|
22
|
+
|
|
23
|
+
const python = ["python3", "python"].find(
|
|
24
|
+
(bin) => spawnSync(bin, ["--version"], { stdio: "ignore" }).status === 0
|
|
25
|
+
);
|
|
26
|
+
if (!python) return; // no interpreter — nothing to do, and nothing to say
|
|
27
|
+
|
|
28
|
+
spawn(python, [script], {
|
|
29
|
+
cwd,
|
|
30
|
+
detached: true,
|
|
31
|
+
stdio: "ignore",
|
|
32
|
+
}).unref();
|
|
33
|
+
} catch {
|
|
34
|
+
// never surface: a mining failure must not affect session start
|
|
35
|
+
}
|
|
36
|
+
}
|
|
6
37
|
|
|
7
38
|
await runSessionStart({
|
|
8
39
|
packageName: "@theglitchking/babel-fish",
|
|
9
40
|
pluginName: "babel-fish",
|
|
10
41
|
configFile: "babel-fish.json",
|
|
11
42
|
});
|
|
43
|
+
|
|
44
|
+
mineVocabulary(process.cwd());
|
package/package.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@theglitchking/babel-fish",
|
|
3
|
-
"version": "2.0
|
|
3
|
+
"version": "2.3.0",
|
|
4
4
|
"description": "Gives your AI coding assistant instant, accurate knowledge of every route, model, service, feature, and infrastructure element in your codebase.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"babel-fish": "bin/babel-fish.js"
|
|
8
8
|
},
|
|
9
9
|
"scripts": {
|
|
10
|
-
"postinstall": "node scripts/link-skills.js"
|
|
10
|
+
"postinstall": "node scripts/link-skills.js",
|
|
11
|
+
"test": "python3 .claude/project-map/test_generate.py && python3 .claude/project-map/test_mine_sessions.py && python3 .claude/project-map/test_grader.py"
|
|
11
12
|
},
|
|
12
13
|
"files": [
|
|
13
14
|
"bin/",
|
|
@@ -15,11 +16,19 @@
|
|
|
15
16
|
"hooks/",
|
|
16
17
|
"scripts/link-skills.js",
|
|
17
18
|
".claude-plugin/",
|
|
18
|
-
".claude/",
|
|
19
|
+
".claude/install.sh",
|
|
20
|
+
".claude/scripts/",
|
|
21
|
+
".claude/templates/",
|
|
22
|
+
".claude/project-map/generate.py",
|
|
23
|
+
".claude/project-map/grader.py",
|
|
24
|
+
".claude/project-map/mine-sessions.py",
|
|
25
|
+
".claude/project-map/test_generate.py",
|
|
26
|
+
".claude/project-map/test_mine_sessions.py",
|
|
27
|
+
".claude/project-map/test_grader.py",
|
|
19
28
|
".githooks/",
|
|
20
29
|
"skills/",
|
|
21
30
|
"install.sh",
|
|
22
|
-
"checksums.json",
|
|
31
|
+
"./checksums.json",
|
|
23
32
|
"LICENSE",
|
|
24
33
|
"README.md",
|
|
25
34
|
"CHANGELOG.md"
|
|
@@ -1,61 +0,0 @@
|
|
|
1
|
-
# babel-fish — Project Map
|
|
2
|
-
|
|
3
|
-
> Auto-generated by `generate.py` on 2026-04-01 20:29. Do not edit manually.
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
## Stats
|
|
7
|
-
|
|
8
|
-
| Metric | Count |
|
|
9
|
-
|--------|-------|
|
|
10
|
-
| API Routes | 0 |
|
|
11
|
-
| Data Models | 0 |
|
|
12
|
-
| Schemas/DTOs | 0 |
|
|
13
|
-
| Frontend Features | 0 |
|
|
14
|
-
| Migrations | 0 |
|
|
15
|
-
| Docker Services | 0 |
|
|
16
|
-
| Vocabulary Entries | 0 |
|
|
17
|
-
| Stack | unknown / unknown |
|
|
18
|
-
|
|
19
|
-
## Section Index
|
|
20
|
-
|
|
21
|
-
| # | Section | Size | When to Read |
|
|
22
|
-
|---|---------|------|--------------|
|
|
23
|
-
| [01](sections/01-vocabulary.md) | 01 Vocabulary | 0.3 KB | Any task — start here if you don't know where the code lives |
|
|
24
|
-
| [02](sections/02-service-topology.md) | 02 Service Topology | 0.1 KB | Debugging connectivity, adding a service, understanding ports |
|
|
25
|
-
| [03](sections/03-environment.md) | 03 Environment | 0.1 KB | Environment setup, missing vars, config issues |
|
|
26
|
-
| [04](sections/04-api-routes.md) | 04 Api Routes | 0.1 KB | Adding/editing API endpoints, checking what routes exist |
|
|
27
|
-
| [05](sections/05-data-models.md) | 05 Data Models | 0.1 KB | Changing database schema, adding fields, understanding relations |
|
|
28
|
-
| [06](sections/06-schemas.md) | 06 Schemas | 0.1 KB | Adding DTOs, changing request/response shapes |
|
|
29
|
-
| [07](sections/07-services.md) | 07 Services | 0.1 KB | Adding service logic, understanding service boundaries |
|
|
30
|
-
| [08](sections/08-background-jobs.md) | 08 Background Jobs | 0.1 KB | Working with background jobs, queues, scheduled tasks |
|
|
31
|
-
| [09](sections/09-frontend-features.md) | 09 Frontend Features | 0.1 KB | Frontend feature work, understanding UI structure |
|
|
32
|
-
| [10](sections/10-tools-commands.md) | 10 Tools Commands | 0.3 KB | Available commands, scripts, developer tooling |
|
|
33
|
-
| [11](sections/11-migrations.md) | 11 Migrations | 0.1 KB | Database migrations, schema history |
|
|
34
|
-
| [12](sections/12-import-chains.md) | 12 Import Chains | 0.2 KB | Tracing data flow from HTTP request to DB |
|
|
35
|
-
| [13](sections/13-frontend-backend-map.md) | 13 Frontend Backend Map | 0.1 KB | Understanding which frontend calls which backend endpoint |
|
|
36
|
-
| [14](sections/14-reverse-proxy.md) | 14 Reverse Proxy | 0.1 KB | Proxy routing, nginx/caddy config |
|
|
37
|
-
| [15](sections/15-auth-config.md) | 15 Auth Config | 0.1 KB | Auth flow, sessions, permissions |
|
|
38
|
-
| [16](sections/16-infra-profile.md) | 16 Infra Profile | 0.3 KB | Infrastructure overview, tech stack summary |
|
|
39
|
-
| [17](sections/17-learned-vocabulary.md) | 17 Learned Vocabulary | 0.2 KB | Vocabulary learned from past sessions |
|
|
40
|
-
| [18](sections/18-dead-code.md) | 18 Dead Code | 0.2 KB | Dead code review |
|
|
41
|
-
| [19](sections/19-doc-pointers.md) | 19 Doc Pointers | 0.1 KB | Finding documentation, READMEs, wikis |
|
|
42
|
-
|
|
43
|
-
## Quick Routing
|
|
44
|
-
|
|
45
|
-
| Task | Read Sections |
|
|
46
|
-
|------|--------------|
|
|
47
|
-
| Feature / UX work | 01 → 09 → 04 |
|
|
48
|
-
| Add model or field | 05 → 06 → 12 |
|
|
49
|
-
| Troubleshoot error | 02 → 03 → 14 |
|
|
50
|
-
| Infrastructure / scaling | 16 → 02 |
|
|
51
|
-
| Auth / security | 15 → 19 |
|
|
52
|
-
| What tools exist | 10 |
|
|
53
|
-
| Background jobs | 08 |
|
|
54
|
-
| Migration history | 11 |
|
|
55
|
-
|
|
56
|
-
## Regenerate
|
|
57
|
-
|
|
58
|
-
```bash
|
|
59
|
-
python .claude/project-map/generate.py # skip if unchanged
|
|
60
|
-
python .claude/project-map/generate.py --force # always regenerate
|
|
61
|
-
```
|
|
Binary file
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
{}
|