@theglitchking/babel-fish 2.0.3 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/.claude/project-map/generate.py +258 -22
  2. package/.claude/project-map/grader.py +77 -1
  3. package/.claude/project-map/mine-sessions.py +141 -23
  4. package/.claude/project-map/test_generate.py +267 -0
  5. package/.claude/project-map/test_grader.py +164 -0
  6. package/.claude/project-map/test_mine_sessions.py +206 -0
  7. package/.claude-plugin/marketplace.json +2 -2
  8. package/.claude-plugin/plugin.json +1 -1
  9. package/.githooks/pre-commit +2 -1
  10. package/CHANGELOG.md +248 -0
  11. package/README.md +9 -2
  12. package/checksums.json +2 -2
  13. package/hooks/session-start.js +34 -1
  14. package/package.json +13 -4
  15. package/.claude/project-map/PROJECT_MAP.md +0 -61
  16. package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
  17. package/.claude/project-map/checksums.json +0 -6
  18. package/.claude/project-map/learned-vocabulary.json +0 -1
  19. package/.claude/project-map/reports/install-report.md +0 -54
  20. package/.claude/project-map/reports/iteration-01-report.md +0 -54
  21. package/.claude/project-map/reports/iteration-01-score.json +0 -7
  22. package/.claude/project-map/sections/01-vocabulary.md +0 -8
  23. package/.claude/project-map/sections/02-service-topology.md +0 -6
  24. package/.claude/project-map/sections/03-environment.md +0 -6
  25. package/.claude/project-map/sections/04-api-routes.md +0 -6
  26. package/.claude/project-map/sections/05-data-models.md +0 -4
  27. package/.claude/project-map/sections/06-schemas.md +0 -4
  28. package/.claude/project-map/sections/07-services.md +0 -6
  29. package/.claude/project-map/sections/08-background-jobs.md +0 -5
  30. package/.claude/project-map/sections/09-frontend-features.md +0 -4
  31. package/.claude/project-map/sections/10-tools-commands.md +0 -8
  32. package/.claude/project-map/sections/11-migrations.md +0 -4
  33. package/.claude/project-map/sections/12-import-chains.md +0 -7
  34. package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
  35. package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
  36. package/.claude/project-map/sections/15-auth-config.md +0 -6
  37. package/.claude/project-map/sections/16-infra-profile.md +0 -13
  38. package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
  39. package/.claude/project-map/sections/18-dead-code.md +0 -9
  40. package/.claude/project-map/sections/19-doc-pointers.md +0 -5
  41. package/.claude/project-map/stack.json +0 -12
  42. package/.claude/rules/operational-runbook.md +0 -40
  43. package/.claude/rules/project-vocabulary.md +0 -25
  44. package/.claude/settings.json +0 -6
  45. package/.claude/settings.local.json +0 -6
  46. package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
@@ -13,10 +13,11 @@ import ast
13
13
  import argparse
14
14
  import hashlib
15
15
  import json
16
+ from fnmatch import fnmatch
16
17
  import re
17
18
  import subprocess
18
19
  import sys
19
- from datetime import datetime
20
+ from datetime import datetime, timezone
20
21
  from pathlib import Path
21
22
  from typing import Any
22
23
 
@@ -33,6 +34,10 @@ MAP_DIR = SCRIPT_DIR
33
34
  SECTIONS_DIR = MAP_DIR / "sections"
34
35
  CHECKSUMS = MAP_DIR / "checksums.json"
35
36
  LEARNED_VOC = MAP_DIR / "learned-vocabulary.json"
37
+ GLOSSARY = MAP_DIR / "glossary.json"
38
+ # Bump when the glossary.json shape changes. Consumers should refuse a major
39
+ # they do not recognise rather than guess.
40
+ GLOSSARY_SCHEMA_VERSION = "1.0"
36
41
 
37
42
  # Resolve project root: two levels up from .claude/project-map/
38
43
  PROJECT_ROOT = MAP_DIR.parent.parent
@@ -53,13 +58,41 @@ def redact_secrets(text: str) -> str:
53
58
  # ── Checksum logic ───────────────────────────────────────────────────────────
54
59
  WATCHED_EXTENSIONS = {
55
60
  '.py', '.ts', '.tsx', '.js', '.jsx', '.go', '.java', '.kt',
56
- '.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env',
61
+ '.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env', '.sh',
57
62
  }
58
63
  WATCHED_NAMES = {
59
64
  'docker-compose.yml', 'docker-compose.yaml', 'docker-compose.dev.yml',
60
65
  'package.json', 'requirements.txt', 'pyproject.toml', 'go.mod',
61
66
  'Cargo.toml', 'pom.xml', 'Gemfile', 'Makefile',
62
67
  }
68
+ # Skill/command manifests — matched by path shape, not by extension, so
69
+ # documentation .md files stay out of the watch set. fnmatch runs against the
70
+ # repo-relative posix path, which anchors the glob at the root:
71
+ # ".documentation/reference/commands/cli.md" does not match "commands/*.md".
72
+ WATCHED_GLOBS = (
73
+ 'skills/*/SKILL.md',
74
+ 'commands/*.md',
75
+ # Dead until '.claude' leaves IGNORE_DIRS; kept as the anchor for when it does.
76
+ '.claude/skills/*/SKILL.md',
77
+ '.claude/commands/*.md',
78
+ )
79
+
80
+ # Directories build_doc_pointers_section() walks. Watched by PATH ONLY (no
81
+ # mtime) so adding or removing a doc refreshes section 19 while editing one
82
+ # does not churn the whole map.
83
+ DOC_DIRS = ['docs', 'doc', 'documentation', 'wiki', '.docs', '.documentation']
84
+ DOC_EXTS = {'.md', '.rst', '.txt', '.adoc'}
85
+ # Auto-generated navigation, not documentation. A hit-em-with-the-docs tree
86
+ # carries one INDEX.md + REGISTRY.md per domain (32 files for 15 domains), which
87
+ # would otherwise crowd every real doc out of section 19's 30-entry cap.
88
+ DOC_SKIP_NAMES = {'INDEX.md', 'REGISTRY.md'}
89
+ # hewtd excludes archive/ from all of its own scans; deprecated docs are not
90
+ # pointers worth handing an agent. 'reports' holds generated, timestamped audit
91
+ # output — listing it is noise, and because every run writes a NEW filename it
92
+ # would change the doc path set and force a full regeneration each time, which
93
+ # is the churn the path-only doc hash exists to prevent.
94
+ DOC_SKIP_DIRS = {'archive', 'reports'}
95
+
63
96
  IGNORE_DIRS = {
64
97
  '.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
65
98
  'dist', 'build', '.next', '.nuxt', 'target', 'vendor', '.cache',
@@ -71,11 +104,34 @@ def collect_watched_files() -> list[Path]:
71
104
  for path in sorted(PROJECT_ROOT.rglob('*')):
72
105
  if any(p in IGNORE_DIRS for p in path.parts):
73
106
  continue
74
- if path.is_file() and (path.suffix in WATCHED_EXTENSIONS or path.name in WATCHED_NAMES):
107
+ if not path.is_file():
108
+ continue
109
+ rel = path.relative_to(PROJECT_ROOT).as_posix()
110
+ if (path.suffix in WATCHED_EXTENSIONS
111
+ or path.name in WATCHED_NAMES
112
+ or any(fnmatch(rel, g) for g in WATCHED_GLOBS)):
75
113
  result.append(path)
76
114
  return result
77
115
 
78
- def compute_checksum(files: list[Path]) -> str:
116
+
117
+ def collect_doc_files() -> list[Path]:
118
+ """Docs section 19 points at. Hashed by path only — see compute_checksum."""
119
+ docs = []
120
+ for doc_dir in DOC_DIRS:
121
+ d = PROJECT_ROOT / doc_dir
122
+ if d.is_dir():
123
+ docs.extend(
124
+ f for f in d.rglob('*')
125
+ if f.is_file()
126
+ and f.suffix in DOC_EXTS
127
+ and f.name not in DOC_SKIP_NAMES
128
+ and not (DOC_SKIP_DIRS & set(f.relative_to(PROJECT_ROOT).parts))
129
+ )
130
+ # Root-level docs, matching build_doc_pointers_section()'s own glob.
131
+ docs.extend(f for f in PROJECT_ROOT.glob('*.md') if f.is_file())
132
+ return sorted(set(docs))
133
+
134
+ def compute_checksum(files: list[Path], doc_files: list[Path] | None = None) -> str:
79
135
  h = hashlib.sha256()
80
136
  for f in files:
81
137
  h.update(str(f).encode())
@@ -83,6 +139,11 @@ def compute_checksum(files: list[Path]) -> str:
83
139
  h.update(str(f.stat().st_mtime_ns).encode())
84
140
  except OSError:
85
141
  pass
142
+ # ponytail: path only, no mtime — a new/renamed/deleted doc regenerates the
143
+ # section 19 pointer list, an edited one doesn't. Hashing doc mtimes would
144
+ # force a full regeneration on every prose edit.
145
+ for f in doc_files or []:
146
+ h.update(str(f).encode())
86
147
  return h.hexdigest()
87
148
 
88
149
  def load_checksums() -> dict:
@@ -710,9 +771,20 @@ class VocabularyBuilder:
710
771
  schemas: list[dict],
711
772
  features: list[dict],
712
773
  stack: dict,
774
+ skills: list[dict] | None = None,
713
775
  ) -> list[dict]:
714
776
  vocab: dict[str, dict] = {}
715
777
 
778
+ # From skills and slash commands. In a plugin repo this is the only
779
+ # source that fires — there are no routes or models to mine.
780
+ for sk in skills or []:
781
+ note = sk['description'][:120] if sk['description'] else f"{sk['kind']} manifest"
782
+ for alias in self._name_to_aliases(sk['name']):
783
+ self._add(vocab, alias, sk['kind'], sk['file'], note)
784
+ if sk['kind'] == 'command':
785
+ # humans say "/status" as often as "status"
786
+ self._add(vocab, f"/{sk['name']}", 'command', sk['file'], note)
787
+
716
788
  # From features
717
789
  for feat in features:
718
790
  name = feat['name']
@@ -781,6 +853,98 @@ class VocabularyBuilder:
781
853
  return {}
782
854
 
783
855
 
856
+ # ── Skill / Command Manifest Parser ───────────────────────────────────────────
857
+
858
+ class SkillParser:
859
+ """Claude skill and slash-command manifests.
860
+
861
+ In a plugin/skill repo these ARE the source: there are no routes or models
862
+ to extract, but a skill's frontmatter name + description is literally an
863
+ alias -> location pair, which is what section 01 wants.
864
+
865
+ The .claude/* patterns are globbed directly, the same way ToolsScanner
866
+ reaches .claude/skills, so a consuming project's installed skills are
867
+ picked up even though '.claude' is in IGNORE_DIRS. Those files are parsed
868
+ but not watched, so they refresh on the next regeneration rather than
869
+ immediately — see WATCHED_GLOBS.
870
+ """
871
+
872
+ PATTERNS = (
873
+ ('skills/*/SKILL.md', 'skill'),
874
+ ('.claude/skills/*/SKILL.md', 'skill'),
875
+ ('commands/*.md', 'command'),
876
+ ('.claude/commands/*.md', 'command'),
877
+ )
878
+
879
+ def parse(self) -> list[dict]:
880
+ found: dict[str, dict] = {}
881
+ for pattern, kind in self.PATTERNS:
882
+ for f in sorted(PROJECT_ROOT.glob(pattern)):
883
+ meta = self._frontmatter(f)
884
+ if meta is None:
885
+ continue
886
+ # commands carry no 'name:' — the filename is the command.
887
+ name = str(meta.get('name') or '').strip()
888
+ if not name:
889
+ name = f.parent.name if f.name == 'SKILL.md' else f.stem
890
+ desc = ' '.join(str(meta.get('description') or '').split())
891
+ rel = str(f.relative_to(PROJECT_ROOT))
892
+ found.setdefault(rel, {
893
+ 'name': name,
894
+ 'kind': kind,
895
+ 'file': rel,
896
+ 'description': desc,
897
+ })
898
+ return list(found.values())
899
+
900
+ def _frontmatter(self, f: Path) -> dict | None:
901
+ """Leading --- ... --- YAML block, or None when absent/unparseable."""
902
+ try:
903
+ text = f.read_text(encoding='utf-8', errors='replace')
904
+ except OSError:
905
+ return None
906
+ if not text.startswith('---'):
907
+ return None
908
+ end = text.find('\n---', 3)
909
+ if end == -1:
910
+ return None
911
+ block = text[3:end]
912
+ if HAS_YAML:
913
+ try:
914
+ data = yaml.safe_load(block)
915
+ if isinstance(data, dict):
916
+ return data
917
+ except Exception:
918
+ pass
919
+ return self._frontmatter_regex(block)
920
+
921
+ def _frontmatter_regex(self, block: str) -> dict:
922
+ """pyyaml-free fallback. Handles 'key: value' and 'key: |' blocks —
923
+ enough for name/description, which is all this parser reads."""
924
+ out: dict[str, str] = {}
925
+ lines = block.splitlines()
926
+ i = 0
927
+ while i < len(lines):
928
+ m = re.match(r'^([A-Za-z_][\w-]*):\s*(.*)$', lines[i])
929
+ if not m:
930
+ i += 1
931
+ continue
932
+ key, val = m.group(1), m.group(2).strip()
933
+ if val in ('|', '>', '|-', '>-', ''):
934
+ # block scalar: consume the indented run beneath it
935
+ body, i = [], i + 1
936
+ while i < len(lines) and (not lines[i].strip() or lines[i][:1] in (' ', '\t')):
937
+ body.append(lines[i].strip())
938
+ i += 1
939
+ joined = ' '.join(x for x in body if x)
940
+ if joined:
941
+ out[key] = joined
942
+ continue
943
+ out[key] = val.strip('"\'')
944
+ i += 1
945
+ return out
946
+
947
+
784
948
  # ── Tools & Commands Scanner ──────────────────────────────────────────────────
785
949
 
786
950
  class ToolsScanner:
@@ -827,6 +991,22 @@ class ToolsScanner:
827
991
  if (skill_dir / 'SKILL.md').exists():
828
992
  tools.append({'name': f'/{skill_dir.name}', 'command': f'/{skill_dir.name}', 'description': 'Claude skill', 'source': 'skills'})
829
993
 
994
+ # Slash commands — commands/*.md and .claude/commands/*.md
995
+ seen = {t['name'] for t in tools}
996
+ for manifest in SkillParser().parse():
997
+ if manifest['kind'] != 'command':
998
+ continue
999
+ name = f"/{manifest['name']}"
1000
+ if name in seen:
1001
+ continue
1002
+ seen.add(name)
1003
+ tools.append({
1004
+ 'name': name,
1005
+ 'command': name,
1006
+ 'description': manifest['description'] or 'slash command',
1007
+ 'source': 'commands',
1008
+ })
1009
+
830
1010
  return tools
831
1011
 
832
1012
 
@@ -958,7 +1138,11 @@ def build_vocabulary_section(vocab: list[dict]) -> str:
958
1138
  notes = v.get('notes', '').replace('|', '\\|')
959
1139
  lines.append(f"| {alias} | {type_} | {location} | {notes} |")
960
1140
  if not vocab:
961
- lines.append("| _(no vocabulary generated yet — add source code to populate)_ | | | |")
1141
+ # Prose, not a table row: a placeholder row parses as a valid entry with
1142
+ # an empty (=neutral) location, which scored an empty vocabulary at 100%
1143
+ # and left grader.py's greenfield branch permanently unreachable.
1144
+ lines.append("")
1145
+ lines.append("_No vocabulary generated yet — add source code to populate._")
962
1146
  return '\n'.join(lines) + '\n'
963
1147
 
964
1148
 
@@ -1260,20 +1444,9 @@ def build_dead_code_section(candidates: list[dict]) -> str:
1260
1444
 
1261
1445
  def build_doc_pointers_section() -> str:
1262
1446
  lines = ["# Section 19 — Documentation Pointers\n\n"]
1263
- docs = []
1264
- doc_dirs = ['docs', 'doc', 'documentation', 'wiki', '.docs']
1265
- doc_exts = {'.md', '.rst', '.txt', '.adoc'}
1266
-
1267
- for doc_dir in doc_dirs:
1268
- d = PROJECT_ROOT / doc_dir
1269
- if d.is_dir():
1270
- for f in sorted(d.rglob('*')):
1271
- if f.is_file() and f.suffix in doc_exts:
1272
- docs.append(str(f.relative_to(PROJECT_ROOT)))
1273
-
1274
- # Root-level docs
1275
- for f in PROJECT_ROOT.glob('*.md'):
1276
- docs.append(str(f.relative_to(PROJECT_ROOT)))
1447
+ # Same walk the checksum watches (DOC_DIRS/DOC_EXTS), so the pointer list and
1448
+ # the watch set cannot drift apart.
1449
+ docs = [str(f.relative_to(PROJECT_ROOT)) for f in collect_doc_files()]
1277
1450
 
1278
1451
  if not docs:
1279
1452
  lines.append("_No documentation files found._\n")
@@ -1283,6 +1456,50 @@ def build_doc_pointers_section() -> str:
1283
1456
  return '\n'.join(lines) + '\n'
1284
1457
 
1285
1458
 
1459
+ # ── Glossary side-channel ─────────────────────────────────────────────────────
1460
+
1461
+ def _relative_source() -> str:
1462
+ vocab_md = SECTIONS_DIR / '01-vocabulary.md'
1463
+ try:
1464
+ return str(vocab_md.relative_to(PROJECT_ROOT))
1465
+ except ValueError:
1466
+ return str(vocab_md)
1467
+
1468
+
1469
+ def write_glossary(vocab: list[dict], stack: dict) -> Path:
1470
+ """Structured vocabulary for machine consumers.
1471
+
1472
+ 01-vocabulary.md stays human-facing. Consumers read this instead of parsing
1473
+ prose: the markdown is a rendered table whose Notes column mixes
1474
+ descriptions with metadata and carries pipe-escaping (`a \\| b`), which is
1475
+ fine to read and hostile to parse.
1476
+ """
1477
+ entries = [
1478
+ {
1479
+ 'key': v.get('alias', ''),
1480
+ 'canonical_path': v.get('location', ''),
1481
+ 'section': v.get('type', ''),
1482
+ 'description': v.get('notes', '') or None,
1483
+ }
1484
+ for v in sorted(vocab, key=lambda x: x.get('alias', ''))
1485
+ if v.get('alias')
1486
+ ]
1487
+ payload = {
1488
+ 'schema_version': GLOSSARY_SCHEMA_VERSION,
1489
+ 'generated_at': datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ'),
1490
+ # SECTIONS_DIR is bound to the script's location, which is not under
1491
+ # PROJECT_ROOT when --project-root points elsewhere. Report a relative
1492
+ # path when there is one, the absolute path otherwise.
1493
+ 'source': _relative_source(),
1494
+ 'project': stack.get('name', PROJECT_ROOT.name),
1495
+ 'entry_count': len(entries),
1496
+ 'entries': entries,
1497
+ }
1498
+ GLOSSARY.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + '\n',
1499
+ encoding='utf-8')
1500
+ return GLOSSARY
1501
+
1502
+
1286
1503
  # ╔══════════════════════════════════════════════════════════════════════════╗
1287
1504
  # ║ PROJECT MAP TOC ║
1288
1505
  # ╚══════════════════════════════════════════════════════════════════════════╝
@@ -1297,6 +1514,7 @@ def build_project_map(
1297
1514
  services: list[dict],
1298
1515
  vocab: list[dict],
1299
1516
  section_files: list[tuple[str, Path]],
1517
+ scanned: dict[str, int] | None = None,
1300
1518
  ) -> str:
1301
1519
  now = datetime.now().strftime('%Y-%m-%d %H:%M')
1302
1520
  name = stack.get('name', PROJECT_ROOT.name)
@@ -1315,6 +1533,16 @@ def build_project_map(
1315
1533
  f"| Docker Services | {len(services)} |",
1316
1534
  f"| Vocabulary Entries | {len(vocab)} |",
1317
1535
  f"| Stack | {stack.get('language','?')} / {stack.get('framework','?')} |",
1536
+ "",
1537
+ # Inputs beside outputs. "0 routes" alone cannot be judged: from 47
1538
+ # Python files it means an extractor is broken, from 0 it is correct.
1539
+ # grader.py reads these to tell those two cases apart.
1540
+ "### Source files scanned\n",
1541
+ "| Language | Files |",
1542
+ "|----------|-------|",
1543
+ ] + [
1544
+ f"| {lang} | {count} |" for lang, count in sorted((scanned or {}).items())
1545
+ ] + [
1318
1546
  "",
1319
1547
  "## Section Index\n",
1320
1548
  "| # | Section | Size | When to Read |",
@@ -1392,7 +1620,7 @@ def main() -> None:
1392
1620
 
1393
1621
  # 1. Checksum check
1394
1622
  watched = collect_watched_files()
1395
- checksum = compute_checksum(watched)
1623
+ checksum = compute_checksum(watched, collect_doc_files())
1396
1624
 
1397
1625
  if not args.force and is_unchanged(checksum):
1398
1626
  print("[generate] ✓ No changes detected — skipping regeneration (use --force to override)")
@@ -1434,12 +1662,16 @@ def main() -> None:
1434
1662
  env_entries = EnvParser().parse()
1435
1663
  migrations = MigrationParser().parse()
1436
1664
  features = FrontendScanner().scan()
1665
+ skills = SkillParser().parse()
1437
1666
  tools = ToolsScanner().scan()
1438
1667
  auth_info = AuthScanner().scan()
1439
1668
  proxy_rules = ReverseProxyScanner().scan()
1440
1669
 
1441
1670
  print("[generate] Building vocabulary...")
1442
- vocab = VocabularyBuilder().build(routes, models, schemas, features, stack)
1671
+ vocab = VocabularyBuilder().build(routes, models, schemas, features, stack, skills)
1672
+
1673
+ glossary_path = write_glossary(vocab, stack)
1674
+ print(f"[generate] Glossary → {glossary_path.name} ({len(vocab)} entries)")
1443
1675
 
1444
1676
  print("[generate] Tracing import chains...")
1445
1677
  chains = ImportChainTracer().trace(routes)
@@ -1473,7 +1705,11 @@ def main() -> None:
1473
1705
 
1474
1706
  # 6. Write PROJECT_MAP.md
1475
1707
  print("[generate] Writing PROJECT_MAP.md...")
1476
- project_map = build_project_map(stack, routes, models, schemas, features, migrations, services, vocab, section_files)
1708
+ project_map = build_project_map(
1709
+ stack, routes, models, schemas, features, migrations, services, vocab, section_files,
1710
+ scanned={'python': len(py_files), 'typescript/javascript': len(ts_files),
1711
+ 'go': len(go_files)},
1712
+ )
1477
1713
  (MAP_DIR / 'PROJECT_MAP.md').write_text(project_map, encoding='utf-8')
1478
1714
 
1479
1715
  # 7. Update checksums
@@ -173,7 +173,7 @@ def grade_import_chains() -> GradeResult:
173
173
 
174
174
  content = chains_file.read_text(encoding='utf-8', errors='replace')
175
175
 
176
- if '_No import chains traced' in content or '_No' in content:
176
+ if '_No import chains traced' in content:
177
177
  # For greenfield or non-Python: acceptable
178
178
  details = "No chains traced (greenfield or non-Python stack — ok)"
179
179
  return GradeResult('import_chain_validity', WEIGHTS['import_chain_validity'],
@@ -497,6 +497,17 @@ def print_terminal_summary(results: list[GradeResult], total_score: float, passe
497
497
  print(f" {'TOTAL SCORE':<30} {score_color}{total_score:5.1f}%{RESET} → {verdict}")
498
498
  print(f"{CYAN}{'─' * 55}{RESET}\n")
499
499
 
500
+ pop, total_sections = populated_sections()
501
+ if total_sections:
502
+ print(f" {'Sections populated':<30} {pop}/{total_sections}"
503
+ f" (diagnostic — not scored)")
504
+ print()
505
+
506
+ for w in usefulness_warnings():
507
+ print(f"{YELLOW} ⚠ {w}{RESET}")
508
+ if usefulness_warnings():
509
+ print()
510
+
500
511
  all_issues = [(r.display_name, i) for r in results for i in r.issues]
501
512
  if all_issues:
502
513
  print(f"{YELLOW} Issues:{RESET}")
@@ -507,6 +518,71 @@ def print_terminal_summary(results: list[GradeResult], total_score: float, passe
507
518
  print()
508
519
 
509
520
 
521
+ # ── Usefulness diagnostics (reported, never scored) ──────────────────────────
522
+ # The seven graded categories all measure FORM, and generate.py always emits
523
+ # well-formed output — so an empty map and a populated one score identically
524
+ # (both 97.0% before this was added). The information that separates them is
525
+ # not in the map, so it is surfaced as a warning rather than folded into the
526
+ # score: making it scored would fail legitimately sparse repos, which is a
527
+ # worse failure than the one it fixes.
528
+
529
+ def _stat(content: str, label: str) -> int | None:
530
+ m = re.search(rf'^\|\s*{re.escape(label)}\s*\|\s*(\d+)\s*\|', content, re.MULTILINE)
531
+ return int(m.group(1)) if m else None
532
+
533
+
534
+ def usefulness_warnings() -> list[str]:
535
+ map_file = MAP_DIR / 'PROJECT_MAP.md'
536
+ if not map_file.exists():
537
+ return []
538
+ content = map_file.read_text(encoding='utf-8', errors='replace')
539
+ warnings = []
540
+
541
+ # Vocabulary is the product. Zero entries means the map gave the user
542
+ # nothing, whatever the seven form categories say. This is the signal that
543
+ # would have surfaced issue #6 at install time.
544
+ vocab = _stat(content, 'Vocabulary Entries')
545
+ if vocab == 0:
546
+ warnings.append(
547
+ "Vocabulary is empty — the map's primary output produced nothing. "
548
+ "Expected on a repo with no source code; otherwise an extractor "
549
+ "does not understand this project's shape."
550
+ )
551
+
552
+ # Source files present but nothing extracted from them: a parser that does
553
+ # not fit the stack, rather than a repo with nothing to find.
554
+ scanned = 0
555
+ block = re.search(r'### Source files scanned(.*?)(?:\n## |\Z)', content, re.DOTALL)
556
+ if block:
557
+ scanned = sum(int(n) for n in re.findall(r'^\|[^|]+\|\s*(\d+)\s*\|',
558
+ block.group(1), re.MULTILINE))
559
+ extracted = sum(v or 0 for v in (
560
+ _stat(content, 'API Routes'), _stat(content, 'Data Models'),
561
+ _stat(content, 'Schemas/DTOs'), _stat(content, 'Frontend Features')))
562
+ if scanned >= 10 and extracted == 0:
563
+ warnings.append(
564
+ f"Scanned {scanned} source file(s) but extracted no routes, models, "
565
+ f"schemas or features — likely an unsupported framework."
566
+ )
567
+ return warnings
568
+
569
+
570
+ def populated_sections() -> tuple[int, int]:
571
+ """Sections carrying real content. Diagnostic only — 19 is aspirational,
572
+ not a target: a plugin repo can never populate routes or migrations."""
573
+ files = sorted(SECTIONS_DIR.glob('*.md')) if SECTIONS_DIR.exists() else []
574
+ pop = 0
575
+ for f in files:
576
+ body = f.read_text(encoding='utf-8', errors='replace')
577
+ meat = [ln for ln in body.splitlines()
578
+ if ln.strip() and not ln.startswith(('#', '>'))
579
+ and not re.fullmatch(r'\|[\s|:-]*\|', ln.strip())
580
+ and not re.match(r'^_\(?(no|none|not)\b', ln.strip(), re.I)]
581
+ if meat:
582
+ pop += 1
583
+ return pop, len(files)
584
+
585
+
510
586
  # ╔══════════════════════════════════════════════════════════════════════════╗
511
587
  # ║ MAIN ║
512
588
  # ╚══════════════════════════════════════════════════════════════════════════╝