@theglitchking/babel-fish 2.0.2 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/.claude/project-map/generate.py +206 -21
  2. package/.claude/project-map/grader.py +77 -1
  3. package/.claude/project-map/mine-sessions.py +141 -23
  4. package/.claude/project-map/test_generate.py +203 -0
  5. package/.claude/project-map/test_grader.py +164 -0
  6. package/.claude/project-map/test_mine_sessions.py +206 -0
  7. package/.claude-plugin/marketplace.json +2 -2
  8. package/.claude-plugin/plugin.json +1 -1
  9. package/CHANGELOG.md +209 -0
  10. package/README.md +9 -2
  11. package/checksums.json +2 -2
  12. package/hooks/session-start.js +34 -1
  13. package/package.json +13 -4
  14. package/.claude/project-map/PROJECT_MAP.md +0 -61
  15. package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
  16. package/.claude/project-map/checksums.json +0 -6
  17. package/.claude/project-map/learned-vocabulary.json +0 -1
  18. package/.claude/project-map/reports/install-report.md +0 -54
  19. package/.claude/project-map/reports/iteration-01-report.md +0 -54
  20. package/.claude/project-map/reports/iteration-01-score.json +0 -7
  21. package/.claude/project-map/sections/01-vocabulary.md +0 -8
  22. package/.claude/project-map/sections/02-service-topology.md +0 -6
  23. package/.claude/project-map/sections/03-environment.md +0 -6
  24. package/.claude/project-map/sections/04-api-routes.md +0 -6
  25. package/.claude/project-map/sections/05-data-models.md +0 -4
  26. package/.claude/project-map/sections/06-schemas.md +0 -4
  27. package/.claude/project-map/sections/07-services.md +0 -6
  28. package/.claude/project-map/sections/08-background-jobs.md +0 -5
  29. package/.claude/project-map/sections/09-frontend-features.md +0 -4
  30. package/.claude/project-map/sections/10-tools-commands.md +0 -8
  31. package/.claude/project-map/sections/11-migrations.md +0 -4
  32. package/.claude/project-map/sections/12-import-chains.md +0 -7
  33. package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
  34. package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
  35. package/.claude/project-map/sections/15-auth-config.md +0 -6
  36. package/.claude/project-map/sections/16-infra-profile.md +0 -13
  37. package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
  38. package/.claude/project-map/sections/18-dead-code.md +0 -9
  39. package/.claude/project-map/sections/19-doc-pointers.md +0 -5
  40. package/.claude/project-map/stack.json +0 -12
  41. package/.claude/rules/operational-runbook.md +0 -40
  42. package/.claude/rules/project-vocabulary.md +0 -25
  43. package/.claude/settings.json +0 -6
  44. package/.claude/settings.local.json +0 -6
  45. package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
@@ -31,6 +31,9 @@ from pathlib import Path
31
31
  # ── Paths ────────────────────────────────────────────────────────────────────
32
32
  SCRIPT_DIR = Path(__file__).parent
33
33
  LEARNED_VOC = SCRIPT_DIR / "learned-vocabulary.json"
34
+ # Incremental cursor. Not merely a cost guard: merge_learned() ADDS scores, so
35
+ # re-mining an already-counted transcript inflates it without bound.
36
+ MINE_CURSOR = SCRIPT_DIR / ".mine-cursor.json"
34
37
  PROJECT_ROOT = SCRIPT_DIR.parent.parent
35
38
 
36
39
  # ── Config ───────────────────────────────────────────────────────────────────
@@ -41,6 +44,7 @@ RECENCY_WINDOWS = [ # (days_threshold, weight)
41
44
  (90, 0.25),
42
45
  ]
43
46
  MAX_ALIAS_WORDS = 6 # Ignore user phrases longer than this
47
+ MAX_LOOKAHEAD = 120 # Messages scanned after a user turn for file activity
44
48
  MIN_ALIAS_WORDS = 1
45
49
  STOP_WORDS = {
46
50
  'the', 'a', 'an', 'this', 'that', 'these', 'those', 'it', 'its',
@@ -61,6 +65,9 @@ FILE_TOOL_NAMES = {
61
65
  'read_file', 'edit_file', 'write_file',
62
66
  }
63
67
 
68
+ # Tools whose arguments are a shell command rather than a file path.
69
+ BASH_TOOL_NAMES = {'Bash', 'bash', 'run_command'}
70
+
64
71
  # ── Recency weight ────────────────────────────────────────────────────────────
65
72
 
66
73
  def recency_weight(session_date: datetime) -> float:
@@ -80,23 +87,27 @@ def find_session_files(project_root: Path) -> list[Path]:
80
87
  """Find Claude Code JSONL session files for this project."""
81
88
  candidates: list[Path] = []
82
89
 
83
- # ~/.claude/projects/ uses a path-encoded slug
84
- # e.g. /mnt/e/the-glitch-kingdom/babel-fish → -mnt-e-the-glitch-kingdom-babel-fish
85
- encoded = str(project_root).replace('/', '-').lstrip('-')
90
+ # ~/.claude/projects/ uses a path-encoded slug that KEEPS the leading
91
+ # separator: /home/u/proj → -home-u-proj. Stripping it (the old
92
+ # .lstrip('-')) meant the exact match never once hit, and every lookup
93
+ # silently fell through to the fuzzy branch below.
94
+ encoded = str(project_root).replace('/', '-')
86
95
  claude_projects = Path.home() / '.claude' / 'projects'
87
96
 
88
97
  if not claude_projects.exists():
89
98
  return []
90
99
 
91
- # Try exact match first
92
100
  exact = claude_projects / encoded
93
101
  if exact.is_dir():
94
- candidates.extend(sorted(exact.glob('*.jsonl')))
102
+ return sorted(exact.glob('*.jsonl'))
95
103
 
96
- # Also try fuzzy match on project name
104
+ # Fallback only when the exact directory is absent. Substring matching on
105
+ # the bare project name is not safe as an addition: "kentro" matches four
106
+ # unrelated project directories, whose aliases would then be attributed to
107
+ # this repo.
97
108
  project_name = project_root.name.lower()
98
- for d in claude_projects.iterdir():
99
- if d.is_dir() and project_name in d.name.lower() and d != exact:
109
+ for d in sorted(claude_projects.iterdir()):
110
+ if d.is_dir() and project_name in d.name.lower():
100
111
  candidates.extend(sorted(d.glob('*.jsonl')))
101
112
 
102
113
  return candidates
@@ -134,14 +145,81 @@ def extract_session_date(messages: list[dict]) -> datetime | None:
134
145
  return None
135
146
 
136
147
 
148
+ # ── JSONL shape helpers ───────────────────────────────────────────────────────
149
+ # Claude Code nests the API message under a "message" key: the role and content
150
+ # live at msg["message"]["role"] / ["content"], not at the top level. Reading the
151
+ # top level yielded 0 tool_use blocks and 0 user messages from a real 1.8 MB
152
+ # transcript (see issue #7). Both shapes are accepted so older or third-party
153
+ # transcripts still parse.
154
+
155
+ def msg_content(msg: dict):
156
+ inner = msg.get('message')
157
+ if isinstance(inner, dict) and inner.get('content') is not None:
158
+ return inner['content']
159
+ return msg.get('content')
160
+
161
+
162
+ def msg_role(msg: dict) -> str:
163
+ inner = msg.get('message')
164
+ if isinstance(inner, dict) and inner.get('role'):
165
+ return str(inner['role'])
166
+ return str(msg.get('role') or msg.get('type') or '')
167
+
168
+
169
+ def is_user_turn(msg: dict) -> bool:
170
+ """A real user message, not a tool_result carrier.
171
+
172
+ Tool results are delivered with role="user": in a real transcript 167 of 181
173
+ role=user messages were tool_result blocks. Treating those as user turns
174
+ made the pairing window close on the assistant's own tool output.
175
+ """
176
+ if msg_role(msg) not in ('user', 'human'):
177
+ return False
178
+ # isMeta marks machine-injected text delivered in the user slot: skill
179
+ # bodies, slash-command definitions, hook output. Mining it learned aliases
180
+ # from the injected docs themselves ("block_index_edits", "superseded by v2
181
+ # guide", "refactor authentication system" — the last from a skill's own
182
+ # example). isSidechain marks subagent transcripts, which are not the user
183
+ # speaking either.
184
+ if msg.get('isMeta') or msg.get('isSidechain'):
185
+ return False
186
+ content = msg_content(msg)
187
+ if isinstance(content, str):
188
+ return bool(content.strip())
189
+ if isinstance(content, list):
190
+ return any(isinstance(b, dict) and b.get('type') == 'text' for b in content)
191
+ return False
192
+
193
+
194
+ # Bash-mediated file access. A real session ran 148 Bash calls against 2 Read
195
+ # and 2 Edit, so a miner that only understands the file tools sees almost
196
+ # nothing. Candidate tokens are only accepted when they resolve to a file that
197
+ # actually exists in the repo — a wrong alias asserted confidently is worse than
198
+ # a missing one, so no cleverer shell parsing than this.
199
+ BASH_TOKEN_RE = re.compile(r'[\w./-]*[\w-]\.[A-Za-z0-9]{1,6}\b')
200
+
201
+
202
+ def extract_paths_from_bash(command: str, project_root: Path) -> list[str]:
203
+ out = []
204
+ for tok in BASH_TOKEN_RE.findall(command or ''):
205
+ tok = tok.strip("'\"()[]{},;:").lstrip('./')
206
+ if not tok or tok.startswith('-'):
207
+ continue
208
+ try:
209
+ if (project_root / tok).is_file():
210
+ out.append(tok)
211
+ except OSError:
212
+ pass
213
+ return out
214
+
215
+
137
216
  # ── Alias extraction ──────────────────────────────────────────────────────────
138
217
 
139
218
  def extract_file_paths_from_tool_calls(messages: list[dict]) -> list[str]:
140
219
  """Extract file paths from tool use calls in a message sequence."""
141
220
  paths = []
142
221
  for msg in messages:
143
- # Handle various JSONL formats
144
- content = msg.get('content') or []
222
+ content = msg_content(msg) or []
145
223
  if isinstance(content, str):
146
224
  continue
147
225
  for block in content:
@@ -150,6 +228,10 @@ def extract_file_paths_from_tool_calls(messages: list[dict]) -> list[str]:
150
228
  if block.get('type') not in ('tool_use', 'tool_result'):
151
229
  continue
152
230
  tool_name = block.get('name', '')
231
+ inp_early = block.get('input') or {}
232
+ if tool_name in BASH_TOOL_NAMES and isinstance(inp_early, dict):
233
+ paths.extend(extract_paths_from_bash(inp_early.get('command', ''), PROJECT_ROOT))
234
+ continue
153
235
  if tool_name not in FILE_TOOL_NAMES:
154
236
  continue
155
237
  # Extract file_path from input
@@ -181,14 +263,14 @@ def extract_user_phrases(text: str) -> list[str]:
181
263
 
182
264
  # "the X" / "the X page/screen/section/tab/view/panel/modal/form/button"
183
265
  for m in re.finditer(
184
- r'\bthe\s+([\w\s-]{2,40?}?)\s*(?:page|screen|section|tab|view|panel|modal|form|button|component|widget|dashboard|list|table|chart|graph|map|sidebar|header|footer|nav|menu)\b',
266
+ r'\bthe\s+([\w\s-]{2,40}?)\s*(?:page|screen|section|tab|view|panel|modal|form|button|component|widget|dashboard|list|table|chart|graph|map|sidebar|header|footer|nav|menu)\b',
185
267
  text, re.IGNORECASE
186
268
  ):
187
269
  phrases.append(m.group(1).lower().strip())
188
270
 
189
271
  # "X feature" / "X functionality" / "X system" / "X module"
190
272
  for m in re.finditer(
191
- r'\b([\w\s-]{2,30?}?)\s+(?:feature|functionality|system|module|service|flow|workflow|pipeline|process)\b',
273
+ r'\b([\w\s-]{2,30}?)\s+(?:feature|functionality|system|module|service|flow|workflow|pipeline|process)\b',
192
274
  text, re.IGNORECASE
193
275
  ):
194
276
  candidate = m.group(1).lower().strip()
@@ -199,7 +281,17 @@ def extract_user_phrases(text: str) -> list[str]:
199
281
  # Filter: remove stop-word-only phrases, too short/long
200
282
  result = []
201
283
  for phrase in phrases:
202
- words = [w for w in phrase.split() if w not in STOP_WORDS and len(w) > 1]
284
+ # Sentence punctuation means this is a clause, not a name for something
285
+ # ("that, if not", "total documents:"). Aliases don't contain it.
286
+ if any(ch in phrase for ch in ',:;?!'):
287
+ continue
288
+ # Strip punctuation BEFORE the stop-word test — otherwise "that," fails
289
+ # to match the stop word "that" and survives as an alias.
290
+ words = [w.strip('.\'"`()[]{}<>*_-') for w in phrase.split()]
291
+ words = [w for w in words if w and w not in STOP_WORDS and len(w) > 1]
292
+ # At least one substantial word, so "up to" and "as is" don't qualify.
293
+ if not any(len(w) >= 3 for w in words):
294
+ continue
203
295
  if MIN_ALIAS_WORDS <= len(words) <= MAX_ALIAS_WORDS:
204
296
  clean = ' '.join(words)
205
297
  if clean and clean not in result:
@@ -241,12 +333,11 @@ class SessionMiner:
241
333
 
242
334
  # Walk messages in pairs: look for user message followed by tool calls
243
335
  for i, msg in enumerate(messages):
244
- role = msg.get('role') or msg.get('type') or ''
245
- if role not in ('user', 'human'):
336
+ if not is_user_turn(msg):
246
337
  continue
247
338
 
248
339
  # Extract text from this user message
249
- content = msg.get('content') or ''
340
+ content = msg_content(msg) or ''
250
341
  if isinstance(content, list):
251
342
  text_parts = []
252
343
  for block in content:
@@ -261,14 +352,13 @@ class SessionMiner:
261
352
  if len(text) < 5:
262
353
  continue
263
354
 
264
- # Find tool calls in subsequent assistant messages (within next 3 messages)
355
+ # Everything the assistant touched before the next real user turn.
356
+ # Bounded by MAX_LOOKAHEAD so one runaway turn can't scan the file.
265
357
  file_paths: list[str] = []
266
- for j in range(i + 1, min(i + 4, len(messages))):
267
- next_msg = messages[j]
268
- next_role = next_msg.get('role') or next_msg.get('type') or ''
269
- if next_role in ('user', 'human') and j > i + 1:
270
- break # New user turn — stop looking
271
- file_paths.extend(extract_file_paths_from_tool_calls([next_msg]))
358
+ for j in range(i + 1, min(i + 1 + MAX_LOOKAHEAD, len(messages))):
359
+ if is_user_turn(messages[j]):
360
+ break
361
+ file_paths.extend(extract_file_paths_from_tool_calls([messages[j]]))
272
362
 
273
363
  if not file_paths:
274
364
  continue
@@ -353,11 +443,27 @@ def decay_old_entries(vocab: dict) -> dict:
353
443
 
354
444
  # ── Main ──────────────────────────────────────────────────────────────────────
355
445
 
446
+ def load_cursor() -> float:
447
+ try:
448
+ return float(json.loads(MINE_CURSOR.read_text()).get('last_mined_mtime', 0.0))
449
+ except Exception:
450
+ return 0.0
451
+
452
+
453
+ def save_cursor(mtime: float) -> None:
454
+ try:
455
+ MINE_CURSOR.write_text(json.dumps({'last_mined_mtime': mtime}, indent=2))
456
+ except OSError:
457
+ pass
458
+
459
+
356
460
  def main() -> None:
357
461
  parser = argparse.ArgumentParser(description='Mine Claude Code sessions for vocabulary aliases')
358
462
  parser.add_argument('--project-root', type=Path, default=None)
359
463
  parser.add_argument('--dry-run', action='store_true', help='Print results without saving')
360
464
  parser.add_argument('--verbose', '-v', action='store_true', help='Show per-file details')
465
+ parser.add_argument('--all', action='store_true',
466
+ help='Ignore the incremental cursor and re-mine every transcript')
361
467
  args = parser.parse_args()
362
468
 
363
469
  global PROJECT_ROOT
@@ -381,6 +487,17 @@ def main() -> None:
381
487
 
382
488
  print(f"[mine-sessions] Found {len(session_files)} session file(s)")
383
489
 
490
+ # Only transcripts touched since the last run. Without this, every session
491
+ # start re-reads the whole history AND re-adds its scores.
492
+ cursor = 0.0 if args.all else load_cursor()
493
+ newest = max((f.stat().st_mtime for f in session_files), default=0.0)
494
+ if cursor:
495
+ session_files = [f for f in session_files if f.stat().st_mtime > cursor]
496
+ if not session_files:
497
+ print("[mine-sessions] No transcripts changed since last run — nothing to do")
498
+ sys.exit(0)
499
+ print(f"[mine-sessions] {len(session_files)} changed since last run")
500
+
384
501
  # Mine
385
502
  miner = SessionMiner(PROJECT_ROOT, verbose=args.verbose)
386
503
  miner.mine(session_files)
@@ -411,6 +528,7 @@ def main() -> None:
411
528
  merged = decay_old_entries(merged)
412
529
 
413
530
  LEARNED_VOC.write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding='utf-8')
531
+ save_cursor(newest)
414
532
  print(f"[mine-sessions] ✓ Saved {len(merged)} total aliases to {LEARNED_VOC}")
415
533
  print(f" (run 'python generate.py --force' to rebuild sections with updated vocabulary)")
416
534
 
@@ -0,0 +1,203 @@
1
+ #!/usr/bin/env python3
2
+ """Regression tests for generate.py — run: python3 .claude/project-map/test_generate.py
3
+
4
+ Stdlib assert + __main__, no pytest, no fixtures (ponytail: the repo has no test
5
+ framework and this doesn't justify adding one).
6
+
7
+ These exist because grader.py cannot catch what they catch: it scored the
8
+ completely empty pre-#6 map at 97.0% PASS, identical to the populated map, since
9
+ "Vocabulary Accuracy" reads 100% on zero entries. The grader measures
10
+ well-formedness; these measure usefulness.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import importlib.util
15
+ import shutil
16
+ import sys
17
+ import tempfile
18
+ from pathlib import Path
19
+
20
+ HERE = Path(__file__).parent
21
+
22
+
23
+ def load_generate(project_root: Path):
24
+ """Fresh module instance bound to project_root."""
25
+ spec = importlib.util.spec_from_file_location("gen_under_test", HERE / "generate.py")
26
+ mod = importlib.util.module_from_spec(spec)
27
+ argv, sys.argv = sys.argv, ["generate.py"]
28
+ try:
29
+ spec.loader.exec_module(mod)
30
+ finally:
31
+ sys.argv = argv
32
+ mod.PROJECT_ROOT = project_root
33
+ return mod
34
+
35
+
36
+ def make_plugin_repo(root: Path) -> None:
37
+ """The repo shape from #6: markdown skills, slash commands, bash, node shim."""
38
+ (root / "skills/demo-skill").mkdir(parents=True)
39
+ (root / "skills/demo-skill/SKILL.md").write_text(
40
+ "---\nname: demo-skill\ndescription: |\n"
41
+ " Does the demo thing.\n Second line of the block scalar.\n---\n\n# Demo\n"
42
+ )
43
+ (root / "commands").mkdir()
44
+ (root / "commands/deploy.md").write_text(
45
+ "---\ndescription: Ship it to production\nallowed-tools: Bash(npx:*)\n---\n\nRun the deploy.\n"
46
+ )
47
+ (root / "install.sh").write_text("#!/usr/bin/env bash\necho hi\n")
48
+ (root / "package.json").write_text('{"name":"demo","scripts":{"build":"tsc"}}\n')
49
+ # A docs tree whose paths must NOT be mistaken for command manifests.
50
+ (root / ".documentation/reference/commands").mkdir(parents=True)
51
+ (root / ".documentation/reference/commands/cli.md").write_text("---\ntitle: CLI\n---\n\nprose\n")
52
+ (root / ".documentation/api").mkdir(parents=True)
53
+ (root / ".documentation/api/contract.md").write_text("---\ntitle: Contract\n---\n\nprose\n")
54
+ (root / ".documentation/api/INDEX.md").write_text("---\ntitle: Index\n---\n\nnav\n")
55
+ (root / ".documentation/reports").mkdir(parents=True)
56
+ (root / ".documentation/reports/maintenance-2026-09-04T18-32-22.md").write_text(
57
+ "---\ntitle: Maintenance Report\n---\n\ngenerated\n"
58
+ )
59
+ (root / ".documentation/archive").mkdir(parents=True)
60
+ (root / ".documentation/archive/old.md").write_text("---\ntitle: Old\n---\n\nretired\n")
61
+ (root / "README.md").write_text("# demo\n")
62
+
63
+
64
+ def checksum(g):
65
+ return g.compute_checksum(g.collect_watched_files(), g.collect_doc_files())
66
+
67
+
68
+ def rel(g, paths):
69
+ return {str(p.relative_to(g.PROJECT_ROOT)) for p in paths}
70
+
71
+
72
+ # ── Watch set ────────────────────────────────────────────────────────────────
73
+
74
+ def test_manifests_are_watched(g):
75
+ watched = rel(g, g.collect_watched_files())
76
+ assert "skills/demo-skill/SKILL.md" in watched, watched
77
+ assert "commands/deploy.md" in watched, watched
78
+ assert "install.sh" in watched, "issue #6: .sh was unwatched, so section 10 never refreshed"
79
+
80
+
81
+ def test_globs_anchor_at_repo_root(g):
82
+ """The whole reason for fnmatch-on-relative-path instead of adding '.md'."""
83
+ watched = rel(g, g.collect_watched_files())
84
+ assert ".documentation/reference/commands/cli.md" not in watched, (
85
+ "a docs tree named commands/ leaked into the watch set — glob is not anchored"
86
+ )
87
+ assert not any(w.startswith(".documentation/") for w in watched), watched
88
+
89
+
90
+ def test_editing_a_skill_moves_the_checksum(g):
91
+ """The #6 blocker: this was bit-identical before the fix."""
92
+ before = checksum(g)
93
+ p = g.PROJECT_ROOT / "skills/demo-skill/SKILL.md"
94
+ p.write_text(p.read_text() + "\nmore\n")
95
+ assert checksum(g) != before, "editing SKILL.md left the checksum unchanged"
96
+
97
+
98
+ # ── Doc pointers ─────────────────────────────────────────────────────────────
99
+
100
+ def test_adding_a_doc_moves_checksum_editing_one_does_not(g):
101
+ before = checksum(g)
102
+ newdoc = g.PROJECT_ROOT / ".documentation/api/added.md"
103
+ newdoc.write_text("---\ntitle: Added\n---\n\nbody\n")
104
+ after_add = checksum(g)
105
+ assert after_add != before, "adding a doc must refresh section 19"
106
+
107
+ newdoc.write_text("---\ntitle: Added\n---\n\nbody, substantially rewritten\n")
108
+ assert checksum(g) == after_add, "editing a doc body must NOT churn the whole map"
109
+
110
+ newdoc.unlink()
111
+ assert checksum(g) == before, "deleting a doc must refresh section 19"
112
+
113
+
114
+ def test_doc_pointers_exclude_nav_and_archive(g):
115
+ docs = rel(g, g.collect_doc_files())
116
+ assert ".documentation/api/contract.md" in docs, docs
117
+ assert "README.md" in docs, "section 19's root *.md glob must be covered"
118
+ assert ".documentation/api/INDEX.md" not in docs, "hewtd nav crowds out real docs"
119
+ assert ".documentation/archive/old.md" not in docs, "archived docs are not pointers"
120
+ assert not any("/reports/" in d for d in docs), (
121
+ "generated reports must not be listed: each run writes a new timestamped "
122
+ "filename, which would move the checksum and force a full regeneration"
123
+ )
124
+
125
+
126
+ # ── SkillParser ──────────────────────────────────────────────────────────────
127
+
128
+ def _parsed(g):
129
+ return {m["name"]: m for m in g.SkillParser().parse()}
130
+
131
+
132
+ def test_skill_and_command_parsed(g):
133
+ got = _parsed(g)
134
+ assert "demo-skill" in got, got
135
+ assert "deploy" in got, "commands/*.md carry no 'name:' — it comes from the filename"
136
+ assert got["demo-skill"]["kind"] == "skill"
137
+ assert got["deploy"]["kind"] == "command"
138
+ assert "Second line" in got["demo-skill"]["description"], "block scalar not joined"
139
+ assert got["deploy"]["description"] == "Ship it to production"
140
+
141
+
142
+ def test_regex_fallback_matches_yaml(g):
143
+ """babel-fish only suggests pyyaml, so the fallback is the common path."""
144
+ if not g.HAS_YAML:
145
+ return # nothing to compare against
146
+ with_yaml = _parsed(g)
147
+ g.HAS_YAML = False
148
+ try:
149
+ without = _parsed(g)
150
+ finally:
151
+ g.HAS_YAML = True
152
+ assert set(with_yaml) == set(without), (set(with_yaml), set(without))
153
+ for k in with_yaml:
154
+ assert with_yaml[k]["description"] == without[k]["description"], k
155
+
156
+
157
+ # ── End to end ───────────────────────────────────────────────────────────────
158
+
159
+ def test_plugin_repo_map_is_not_empty(g):
160
+ """The test that fails if #6 regresses."""
161
+ skills = g.SkillParser().parse()
162
+ vocab = g.VocabularyBuilder().build([], [], [], [], g.load_stack(), skills)
163
+ aliases = {v["alias"] for v in vocab}
164
+ assert aliases, "a plugin repo produced an empty vocabulary — issue #6"
165
+ assert "demo skill" in aliases or "demo-skill" in aliases, aliases
166
+ assert "/deploy" in aliases, "commands should be reachable as /name"
167
+ assert "deploy" in aliases, "...and as a bare word"
168
+
169
+ tools = {t["name"] for t in g.ToolsScanner().scan()}
170
+ assert "/deploy" in tools, tools
171
+
172
+ section = g.build_vocabulary_section(vocab)
173
+ assert "no vocabulary generated yet" not in section, "section 01 still renders the stub"
174
+
175
+
176
+ def main() -> int:
177
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
178
+ failed = []
179
+ tmp = Path(tempfile.mkdtemp(prefix="babelfish-test-"))
180
+ try:
181
+ for fn in tests:
182
+ root = tmp / fn.__name__
183
+ root.mkdir()
184
+ make_plugin_repo(root)
185
+ g = load_generate(root)
186
+ try:
187
+ fn(g)
188
+ print(f" ok {fn.__name__}")
189
+ except Exception as e:
190
+ # Exception, not just AssertionError: against an older
191
+ # generate.py the new helpers are simply absent, and that
192
+ # should read as a failing test, not abort the whole run.
193
+ failed.append((fn.__name__, e))
194
+ print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
195
+ finally:
196
+ shutil.rmtree(tmp, ignore_errors=True)
197
+
198
+ print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
199
+ return 1 if failed else 0
200
+
201
+
202
+ if __name__ == "__main__":
203
+ sys.exit(main())
@@ -0,0 +1,164 @@
1
+ #!/usr/bin/env python3
2
+ """Regression tests for grader.py — run: python3 .claude/project-map/test_grader.py
3
+
4
+ Guards the fixes from issue #9. The headline defect there was that the grader
5
+ scored a completely empty map and a populated one identically (97.0% both), so
6
+ these assert the SIGNAL that separates them, not the score — deliberately, since
7
+ folding usefulness into the weighted score would fail legitimately sparse repos.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import importlib.util
12
+ import shutil
13
+ import sys
14
+ import tempfile
15
+ from pathlib import Path
16
+
17
+ HERE = Path(__file__).parent
18
+
19
+
20
+ def load_grader(map_dir: Path):
21
+ spec = importlib.util.spec_from_file_location("grader_under_test", HERE / "grader.py")
22
+ g = importlib.util.module_from_spec(spec)
23
+ argv, sys.argv = sys.argv, ["grader.py"]
24
+ try:
25
+ spec.loader.exec_module(g)
26
+ finally:
27
+ sys.argv = argv
28
+ g.MAP_DIR = map_dir
29
+ g.SECTIONS_DIR = map_dir / "sections"
30
+ g.PROJECT_ROOT = map_dir
31
+ return g
32
+
33
+
34
+ def write_map(map_dir: Path, *, vocab: int, scanned: int, extracted: int = 0) -> None:
35
+ (map_dir / "sections").mkdir(parents=True, exist_ok=True)
36
+ (map_dir / "PROJECT_MAP.md").write_text(f"""## Stats
37
+
38
+ | Metric | Count |
39
+ |--------|-------|
40
+ | API Routes | {extracted} |
41
+ | Data Models | 0 |
42
+ | Schemas/DTOs | 0 |
43
+ | Frontend Features | 0 |
44
+ | Vocabulary Entries | {vocab} |
45
+
46
+ ### Source files scanned
47
+
48
+ | Language | Files |
49
+ |----------|-------|
50
+ | python | {scanned} |
51
+
52
+ ## Section Index
53
+ """)
54
+
55
+
56
+ def test_empty_vocabulary_warns(tmp):
57
+ write_map(tmp, vocab=0, scanned=0)
58
+ w = load_grader(tmp).usefulness_warnings()
59
+ assert any("Vocabulary is empty" in x for x in w), w
60
+
61
+
62
+ def test_populated_vocabulary_does_not_warn(tmp):
63
+ write_map(tmp, vocab=10, scanned=3)
64
+ w = load_grader(tmp).usefulness_warnings()
65
+ assert not any("Vocabulary is empty" in x for x in w), w
66
+
67
+
68
+ def test_many_files_no_extraction_warns(tmp):
69
+ write_map(tmp, vocab=7, scanned=47, extracted=0)
70
+ w = load_grader(tmp).usefulness_warnings()
71
+ assert any("Scanned 47" in x for x in w), w
72
+
73
+
74
+ def test_few_files_no_extraction_is_quiet(tmp):
75
+ """This repo: 3 CLI shims yielding no routes is correct, not a defect."""
76
+ write_map(tmp, vocab=10, scanned=3, extracted=0)
77
+ w = load_grader(tmp).usefulness_warnings()
78
+ assert not any("Scanned" in x for x in w), w
79
+
80
+
81
+ def test_warnings_do_not_touch_the_score(tmp):
82
+ """Usefulness is reported, never scored — scoring it would fail sparse repos."""
83
+ g = load_grader(tmp)
84
+ write_map(tmp, vocab=0, scanned=99)
85
+ for name in dir(g):
86
+ if name.startswith("grade_"):
87
+ r = getattr(g, name)()
88
+ assert 0.0 <= r.raw_score <= 100.0
89
+ assert g.usefulness_warnings(), "expected warnings for an empty map"
90
+
91
+
92
+ def test_populated_chains_not_misread_as_empty(tmp):
93
+ """`or '_No' in content` matched _Note / _Nothing anywhere in the file and
94
+ scored a populated section as an acceptable empty one at a flat 80%."""
95
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
96
+ (tmp / "sections" / "12-import-chains.md").write_text(
97
+ "# Section 12\n\n_Note: partial._\n\n```\nsrc/a.py -> src/b.py\n```\n")
98
+ g = load_grader(tmp)
99
+ r = g.grade_import_chains()
100
+ assert "No chains traced" not in r.details, (
101
+ f"a populated section was scored as empty: {r.details}")
102
+
103
+
104
+ def test_stub_chains_still_recognised(tmp):
105
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
106
+ (tmp / "sections" / "12-import-chains.md").write_text(
107
+ "# Section 12\n\n_No import chains traced._\n")
108
+ r = load_grader(tmp).grade_import_chains()
109
+ assert "No chains traced" in r.details, r.details
110
+
111
+
112
+ def test_empty_vocab_section_reaches_greenfield_branch(tmp):
113
+ """generate.py used to emit a placeholder TABLE ROW, which parsed as a valid
114
+ entry with a neutral location — scoring an empty vocabulary 100% and leaving
115
+ this branch permanently unreachable."""
116
+ (tmp / "sections").mkdir(parents=True, exist_ok=True)
117
+ (tmp / "sections" / "01-vocabulary.md").write_text(
118
+ "# Section 01\n\n| Alias | Type | Location | Notes |\n|---|---|---|---|\n\n"
119
+ "_No vocabulary generated yet — add source code to populate._\n")
120
+ r = load_grader(tmp).grade_vocabulary_accuracy()
121
+ assert r.raw_score == 85.0, f"greenfield branch not reached: {r.raw_score} {r.details}"
122
+
123
+
124
+ def test_generate_emits_prose_not_a_row_for_empty_vocab():
125
+ spec = importlib.util.spec_from_file_location("gen", HERE / "generate.py")
126
+ gen = importlib.util.module_from_spec(spec)
127
+ argv, sys.argv = sys.argv, ["generate.py"]
128
+ try:
129
+ spec.loader.exec_module(gen)
130
+ finally:
131
+ sys.argv = argv
132
+ out = gen.build_vocabulary_section([])
133
+ rows = [l for l in out.splitlines()
134
+ if l.strip().startswith("|") and "Alias" not in l and set(l.strip()) - set("|-: ")]
135
+ assert not rows, f"empty vocabulary must not emit a table row, got {rows}"
136
+ assert "No vocabulary generated yet" in out
137
+
138
+
139
+ def main() -> int:
140
+ import inspect
141
+ tests = [v for k, v in sorted(globals().items()) if k.startswith("test_")]
142
+ failed = []
143
+ base = Path(tempfile.mkdtemp(prefix="grader-test-"))
144
+ try:
145
+ for fn in tests:
146
+ d = base / fn.__name__
147
+ d.mkdir(parents=True)
148
+ try:
149
+ if "tmp" in inspect.signature(fn).parameters:
150
+ fn(d)
151
+ else:
152
+ fn()
153
+ print(f" ok {fn.__name__}")
154
+ except Exception as e:
155
+ failed.append(fn.__name__)
156
+ print(f" FAIL {fn.__name__}: {type(e).__name__}: {e}")
157
+ finally:
158
+ shutil.rmtree(base, ignore_errors=True)
159
+ print(f"\n{len(tests) - len(failed)}/{len(tests)} passed")
160
+ return 1 if failed else 0
161
+
162
+
163
+ if __name__ == "__main__":
164
+ sys.exit(main())