@theglitchking/babel-fish 2.0.3 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/project-map/generate.py +258 -22
- package/.claude/project-map/grader.py +77 -1
- package/.claude/project-map/mine-sessions.py +141 -23
- package/.claude/project-map/test_generate.py +267 -0
- package/.claude/project-map/test_grader.py +164 -0
- package/.claude/project-map/test_mine_sessions.py +206 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/.claude-plugin/plugin.json +1 -1
- package/.githooks/pre-commit +2 -1
- package/CHANGELOG.md +248 -0
- package/README.md +9 -2
- package/checksums.json +2 -2
- package/hooks/session-start.js +34 -1
- package/package.json +13 -4
- package/.claude/project-map/PROJECT_MAP.md +0 -61
- package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
- package/.claude/project-map/checksums.json +0 -6
- package/.claude/project-map/learned-vocabulary.json +0 -1
- package/.claude/project-map/reports/install-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-report.md +0 -54
- package/.claude/project-map/reports/iteration-01-score.json +0 -7
- package/.claude/project-map/sections/01-vocabulary.md +0 -8
- package/.claude/project-map/sections/02-service-topology.md +0 -6
- package/.claude/project-map/sections/03-environment.md +0 -6
- package/.claude/project-map/sections/04-api-routes.md +0 -6
- package/.claude/project-map/sections/05-data-models.md +0 -4
- package/.claude/project-map/sections/06-schemas.md +0 -4
- package/.claude/project-map/sections/07-services.md +0 -6
- package/.claude/project-map/sections/08-background-jobs.md +0 -5
- package/.claude/project-map/sections/09-frontend-features.md +0 -4
- package/.claude/project-map/sections/10-tools-commands.md +0 -8
- package/.claude/project-map/sections/11-migrations.md +0 -4
- package/.claude/project-map/sections/12-import-chains.md +0 -7
- package/.claude/project-map/sections/13-frontend-backend-map.md +0 -8
- package/.claude/project-map/sections/14-reverse-proxy.md +0 -4
- package/.claude/project-map/sections/15-auth-config.md +0 -6
- package/.claude/project-map/sections/16-infra-profile.md +0 -13
- package/.claude/project-map/sections/17-learned-vocabulary.md +0 -7
- package/.claude/project-map/sections/18-dead-code.md +0 -9
- package/.claude/project-map/sections/19-doc-pointers.md +0 -5
- package/.claude/project-map/stack.json +0 -12
- package/.claude/rules/operational-runbook.md +0 -40
- package/.claude/rules/project-vocabulary.md +0 -25
- package/.claude/settings.json +0 -6
- package/.claude/settings.local.json +0 -6
- package/.claude/skills/babel-fish-developer-skill/SKILL.md +0 -56
|
@@ -13,10 +13,11 @@ import ast
|
|
|
13
13
|
import argparse
|
|
14
14
|
import hashlib
|
|
15
15
|
import json
|
|
16
|
+
from fnmatch import fnmatch
|
|
16
17
|
import re
|
|
17
18
|
import subprocess
|
|
18
19
|
import sys
|
|
19
|
-
from datetime import datetime
|
|
20
|
+
from datetime import datetime, timezone
|
|
20
21
|
from pathlib import Path
|
|
21
22
|
from typing import Any
|
|
22
23
|
|
|
@@ -33,6 +34,10 @@ MAP_DIR = SCRIPT_DIR
|
|
|
33
34
|
SECTIONS_DIR = MAP_DIR / "sections"
|
|
34
35
|
CHECKSUMS = MAP_DIR / "checksums.json"
|
|
35
36
|
LEARNED_VOC = MAP_DIR / "learned-vocabulary.json"
|
|
37
|
+
GLOSSARY = MAP_DIR / "glossary.json"
|
|
38
|
+
# Bump when the glossary.json shape changes. Consumers should refuse a major
|
|
39
|
+
# they do not recognise rather than guess.
|
|
40
|
+
GLOSSARY_SCHEMA_VERSION = "1.0"
|
|
36
41
|
|
|
37
42
|
# Resolve project root: two levels up from .claude/project-map/
|
|
38
43
|
PROJECT_ROOT = MAP_DIR.parent.parent
|
|
@@ -53,13 +58,41 @@ def redact_secrets(text: str) -> str:
|
|
|
53
58
|
# ── Checksum logic ───────────────────────────────────────────────────────────
|
|
54
59
|
WATCHED_EXTENSIONS = {
|
|
55
60
|
'.py', '.ts', '.tsx', '.js', '.jsx', '.go', '.java', '.kt',
|
|
56
|
-
'.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env',
|
|
61
|
+
'.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env', '.sh',
|
|
57
62
|
}
|
|
58
63
|
WATCHED_NAMES = {
|
|
59
64
|
'docker-compose.yml', 'docker-compose.yaml', 'docker-compose.dev.yml',
|
|
60
65
|
'package.json', 'requirements.txt', 'pyproject.toml', 'go.mod',
|
|
61
66
|
'Cargo.toml', 'pom.xml', 'Gemfile', 'Makefile',
|
|
62
67
|
}
|
|
68
|
+
# Skill/command manifests — matched by path shape, not by extension, so
|
|
69
|
+
# documentation .md files stay out of the watch set. fnmatch runs against the
|
|
70
|
+
# repo-relative posix path, which anchors the glob at the root:
|
|
71
|
+
# ".documentation/reference/commands/cli.md" does not match "commands/*.md".
|
|
72
|
+
WATCHED_GLOBS = (
|
|
73
|
+
'skills/*/SKILL.md',
|
|
74
|
+
'commands/*.md',
|
|
75
|
+
# Dead until '.claude' leaves IGNORE_DIRS; kept as the anchor for when it does.
|
|
76
|
+
'.claude/skills/*/SKILL.md',
|
|
77
|
+
'.claude/commands/*.md',
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
# Directories build_doc_pointers_section() walks. Watched by PATH ONLY (no
|
|
81
|
+
# mtime) so adding or removing a doc refreshes section 19 while editing one
|
|
82
|
+
# does not churn the whole map.
|
|
83
|
+
DOC_DIRS = ['docs', 'doc', 'documentation', 'wiki', '.docs', '.documentation']
|
|
84
|
+
DOC_EXTS = {'.md', '.rst', '.txt', '.adoc'}
|
|
85
|
+
# Auto-generated navigation, not documentation. A hit-em-with-the-docs tree
|
|
86
|
+
# carries one INDEX.md + REGISTRY.md per domain (32 files for 15 domains), which
|
|
87
|
+
# would otherwise crowd every real doc out of section 19's 30-entry cap.
|
|
88
|
+
DOC_SKIP_NAMES = {'INDEX.md', 'REGISTRY.md'}
|
|
89
|
+
# hewtd excludes archive/ from all of its own scans; deprecated docs are not
|
|
90
|
+
# pointers worth handing an agent. 'reports' holds generated, timestamped audit
|
|
91
|
+
# output — listing it is noise, and because every run writes a NEW filename it
|
|
92
|
+
# would change the doc path set and force a full regeneration each time, which
|
|
93
|
+
# is the churn the path-only doc hash exists to prevent.
|
|
94
|
+
DOC_SKIP_DIRS = {'archive', 'reports'}
|
|
95
|
+
|
|
63
96
|
IGNORE_DIRS = {
|
|
64
97
|
'.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
|
|
65
98
|
'dist', 'build', '.next', '.nuxt', 'target', 'vendor', '.cache',
|
|
@@ -71,11 +104,34 @@ def collect_watched_files() -> list[Path]:
|
|
|
71
104
|
for path in sorted(PROJECT_ROOT.rglob('*')):
|
|
72
105
|
if any(p in IGNORE_DIRS for p in path.parts):
|
|
73
106
|
continue
|
|
74
|
-
if path.is_file()
|
|
107
|
+
if not path.is_file():
|
|
108
|
+
continue
|
|
109
|
+
rel = path.relative_to(PROJECT_ROOT).as_posix()
|
|
110
|
+
if (path.suffix in WATCHED_EXTENSIONS
|
|
111
|
+
or path.name in WATCHED_NAMES
|
|
112
|
+
or any(fnmatch(rel, g) for g in WATCHED_GLOBS)):
|
|
75
113
|
result.append(path)
|
|
76
114
|
return result
|
|
77
115
|
|
|
78
|
-
|
|
116
|
+
|
|
117
|
+
def collect_doc_files() -> list[Path]:
|
|
118
|
+
"""Docs section 19 points at. Hashed by path only — see compute_checksum."""
|
|
119
|
+
docs = []
|
|
120
|
+
for doc_dir in DOC_DIRS:
|
|
121
|
+
d = PROJECT_ROOT / doc_dir
|
|
122
|
+
if d.is_dir():
|
|
123
|
+
docs.extend(
|
|
124
|
+
f for f in d.rglob('*')
|
|
125
|
+
if f.is_file()
|
|
126
|
+
and f.suffix in DOC_EXTS
|
|
127
|
+
and f.name not in DOC_SKIP_NAMES
|
|
128
|
+
and not (DOC_SKIP_DIRS & set(f.relative_to(PROJECT_ROOT).parts))
|
|
129
|
+
)
|
|
130
|
+
# Root-level docs, matching build_doc_pointers_section()'s own glob.
|
|
131
|
+
docs.extend(f for f in PROJECT_ROOT.glob('*.md') if f.is_file())
|
|
132
|
+
return sorted(set(docs))
|
|
133
|
+
|
|
134
|
+
def compute_checksum(files: list[Path], doc_files: list[Path] | None = None) -> str:
|
|
79
135
|
h = hashlib.sha256()
|
|
80
136
|
for f in files:
|
|
81
137
|
h.update(str(f).encode())
|
|
@@ -83,6 +139,11 @@ def compute_checksum(files: list[Path]) -> str:
|
|
|
83
139
|
h.update(str(f.stat().st_mtime_ns).encode())
|
|
84
140
|
except OSError:
|
|
85
141
|
pass
|
|
142
|
+
# ponytail: path only, no mtime — a new/renamed/deleted doc regenerates the
|
|
143
|
+
# section 19 pointer list, an edited one doesn't. Hashing doc mtimes would
|
|
144
|
+
# force a full regeneration on every prose edit.
|
|
145
|
+
for f in doc_files or []:
|
|
146
|
+
h.update(str(f).encode())
|
|
86
147
|
return h.hexdigest()
|
|
87
148
|
|
|
88
149
|
def load_checksums() -> dict:
|
|
@@ -710,9 +771,20 @@ class VocabularyBuilder:
|
|
|
710
771
|
schemas: list[dict],
|
|
711
772
|
features: list[dict],
|
|
712
773
|
stack: dict,
|
|
774
|
+
skills: list[dict] | None = None,
|
|
713
775
|
) -> list[dict]:
|
|
714
776
|
vocab: dict[str, dict] = {}
|
|
715
777
|
|
|
778
|
+
# From skills and slash commands. In a plugin repo this is the only
|
|
779
|
+
# source that fires — there are no routes or models to mine.
|
|
780
|
+
for sk in skills or []:
|
|
781
|
+
note = sk['description'][:120] if sk['description'] else f"{sk['kind']} manifest"
|
|
782
|
+
for alias in self._name_to_aliases(sk['name']):
|
|
783
|
+
self._add(vocab, alias, sk['kind'], sk['file'], note)
|
|
784
|
+
if sk['kind'] == 'command':
|
|
785
|
+
# humans say "/status" as often as "status"
|
|
786
|
+
self._add(vocab, f"/{sk['name']}", 'command', sk['file'], note)
|
|
787
|
+
|
|
716
788
|
# From features
|
|
717
789
|
for feat in features:
|
|
718
790
|
name = feat['name']
|
|
@@ -781,6 +853,98 @@ class VocabularyBuilder:
|
|
|
781
853
|
return {}
|
|
782
854
|
|
|
783
855
|
|
|
856
|
+
# ── Skill / Command Manifest Parser ───────────────────────────────────────────
|
|
857
|
+
|
|
858
|
+
class SkillParser:
|
|
859
|
+
"""Claude skill and slash-command manifests.
|
|
860
|
+
|
|
861
|
+
In a plugin/skill repo these ARE the source: there are no routes or models
|
|
862
|
+
to extract, but a skill's frontmatter name + description is literally an
|
|
863
|
+
alias -> location pair, which is what section 01 wants.
|
|
864
|
+
|
|
865
|
+
The .claude/* patterns are globbed directly, the same way ToolsScanner
|
|
866
|
+
reaches .claude/skills, so a consuming project's installed skills are
|
|
867
|
+
picked up even though '.claude' is in IGNORE_DIRS. Those files are parsed
|
|
868
|
+
but not watched, so they refresh on the next regeneration rather than
|
|
869
|
+
immediately — see WATCHED_GLOBS.
|
|
870
|
+
"""
|
|
871
|
+
|
|
872
|
+
PATTERNS = (
|
|
873
|
+
('skills/*/SKILL.md', 'skill'),
|
|
874
|
+
('.claude/skills/*/SKILL.md', 'skill'),
|
|
875
|
+
('commands/*.md', 'command'),
|
|
876
|
+
('.claude/commands/*.md', 'command'),
|
|
877
|
+
)
|
|
878
|
+
|
|
879
|
+
def parse(self) -> list[dict]:
|
|
880
|
+
found: dict[str, dict] = {}
|
|
881
|
+
for pattern, kind in self.PATTERNS:
|
|
882
|
+
for f in sorted(PROJECT_ROOT.glob(pattern)):
|
|
883
|
+
meta = self._frontmatter(f)
|
|
884
|
+
if meta is None:
|
|
885
|
+
continue
|
|
886
|
+
# commands carry no 'name:' — the filename is the command.
|
|
887
|
+
name = str(meta.get('name') or '').strip()
|
|
888
|
+
if not name:
|
|
889
|
+
name = f.parent.name if f.name == 'SKILL.md' else f.stem
|
|
890
|
+
desc = ' '.join(str(meta.get('description') or '').split())
|
|
891
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
892
|
+
found.setdefault(rel, {
|
|
893
|
+
'name': name,
|
|
894
|
+
'kind': kind,
|
|
895
|
+
'file': rel,
|
|
896
|
+
'description': desc,
|
|
897
|
+
})
|
|
898
|
+
return list(found.values())
|
|
899
|
+
|
|
900
|
+
def _frontmatter(self, f: Path) -> dict | None:
|
|
901
|
+
"""Leading --- ... --- YAML block, or None when absent/unparseable."""
|
|
902
|
+
try:
|
|
903
|
+
text = f.read_text(encoding='utf-8', errors='replace')
|
|
904
|
+
except OSError:
|
|
905
|
+
return None
|
|
906
|
+
if not text.startswith('---'):
|
|
907
|
+
return None
|
|
908
|
+
end = text.find('\n---', 3)
|
|
909
|
+
if end == -1:
|
|
910
|
+
return None
|
|
911
|
+
block = text[3:end]
|
|
912
|
+
if HAS_YAML:
|
|
913
|
+
try:
|
|
914
|
+
data = yaml.safe_load(block)
|
|
915
|
+
if isinstance(data, dict):
|
|
916
|
+
return data
|
|
917
|
+
except Exception:
|
|
918
|
+
pass
|
|
919
|
+
return self._frontmatter_regex(block)
|
|
920
|
+
|
|
921
|
+
def _frontmatter_regex(self, block: str) -> dict:
|
|
922
|
+
"""pyyaml-free fallback. Handles 'key: value' and 'key: |' blocks —
|
|
923
|
+
enough for name/description, which is all this parser reads."""
|
|
924
|
+
out: dict[str, str] = {}
|
|
925
|
+
lines = block.splitlines()
|
|
926
|
+
i = 0
|
|
927
|
+
while i < len(lines):
|
|
928
|
+
m = re.match(r'^([A-Za-z_][\w-]*):\s*(.*)$', lines[i])
|
|
929
|
+
if not m:
|
|
930
|
+
i += 1
|
|
931
|
+
continue
|
|
932
|
+
key, val = m.group(1), m.group(2).strip()
|
|
933
|
+
if val in ('|', '>', '|-', '>-', ''):
|
|
934
|
+
# block scalar: consume the indented run beneath it
|
|
935
|
+
body, i = [], i + 1
|
|
936
|
+
while i < len(lines) and (not lines[i].strip() or lines[i][:1] in (' ', '\t')):
|
|
937
|
+
body.append(lines[i].strip())
|
|
938
|
+
i += 1
|
|
939
|
+
joined = ' '.join(x for x in body if x)
|
|
940
|
+
if joined:
|
|
941
|
+
out[key] = joined
|
|
942
|
+
continue
|
|
943
|
+
out[key] = val.strip('"\'')
|
|
944
|
+
i += 1
|
|
945
|
+
return out
|
|
946
|
+
|
|
947
|
+
|
|
784
948
|
# ── Tools & Commands Scanner ──────────────────────────────────────────────────
|
|
785
949
|
|
|
786
950
|
class ToolsScanner:
|
|
@@ -827,6 +991,22 @@ class ToolsScanner:
|
|
|
827
991
|
if (skill_dir / 'SKILL.md').exists():
|
|
828
992
|
tools.append({'name': f'/{skill_dir.name}', 'command': f'/{skill_dir.name}', 'description': 'Claude skill', 'source': 'skills'})
|
|
829
993
|
|
|
994
|
+
# Slash commands — commands/*.md and .claude/commands/*.md
|
|
995
|
+
seen = {t['name'] for t in tools}
|
|
996
|
+
for manifest in SkillParser().parse():
|
|
997
|
+
if manifest['kind'] != 'command':
|
|
998
|
+
continue
|
|
999
|
+
name = f"/{manifest['name']}"
|
|
1000
|
+
if name in seen:
|
|
1001
|
+
continue
|
|
1002
|
+
seen.add(name)
|
|
1003
|
+
tools.append({
|
|
1004
|
+
'name': name,
|
|
1005
|
+
'command': name,
|
|
1006
|
+
'description': manifest['description'] or 'slash command',
|
|
1007
|
+
'source': 'commands',
|
|
1008
|
+
})
|
|
1009
|
+
|
|
830
1010
|
return tools
|
|
831
1011
|
|
|
832
1012
|
|
|
@@ -958,7 +1138,11 @@ def build_vocabulary_section(vocab: list[dict]) -> str:
|
|
|
958
1138
|
notes = v.get('notes', '').replace('|', '\\|')
|
|
959
1139
|
lines.append(f"| {alias} | {type_} | {location} | {notes} |")
|
|
960
1140
|
if not vocab:
|
|
961
|
-
|
|
1141
|
+
# Prose, not a table row: a placeholder row parses as a valid entry with
|
|
1142
|
+
# an empty (=neutral) location, which scored an empty vocabulary at 100%
|
|
1143
|
+
# and left grader.py's greenfield branch permanently unreachable.
|
|
1144
|
+
lines.append("")
|
|
1145
|
+
lines.append("_No vocabulary generated yet — add source code to populate._")
|
|
962
1146
|
return '\n'.join(lines) + '\n'
|
|
963
1147
|
|
|
964
1148
|
|
|
@@ -1260,20 +1444,9 @@ def build_dead_code_section(candidates: list[dict]) -> str:
|
|
|
1260
1444
|
|
|
1261
1445
|
def build_doc_pointers_section() -> str:
|
|
1262
1446
|
lines = ["# Section 19 — Documentation Pointers\n\n"]
|
|
1263
|
-
|
|
1264
|
-
|
|
1265
|
-
|
|
1266
|
-
|
|
1267
|
-
for doc_dir in doc_dirs:
|
|
1268
|
-
d = PROJECT_ROOT / doc_dir
|
|
1269
|
-
if d.is_dir():
|
|
1270
|
-
for f in sorted(d.rglob('*')):
|
|
1271
|
-
if f.is_file() and f.suffix in doc_exts:
|
|
1272
|
-
docs.append(str(f.relative_to(PROJECT_ROOT)))
|
|
1273
|
-
|
|
1274
|
-
# Root-level docs
|
|
1275
|
-
for f in PROJECT_ROOT.glob('*.md'):
|
|
1276
|
-
docs.append(str(f.relative_to(PROJECT_ROOT)))
|
|
1447
|
+
# Same walk the checksum watches (DOC_DIRS/DOC_EXTS), so the pointer list and
|
|
1448
|
+
# the watch set cannot drift apart.
|
|
1449
|
+
docs = [str(f.relative_to(PROJECT_ROOT)) for f in collect_doc_files()]
|
|
1277
1450
|
|
|
1278
1451
|
if not docs:
|
|
1279
1452
|
lines.append("_No documentation files found._\n")
|
|
@@ -1283,6 +1456,50 @@ def build_doc_pointers_section() -> str:
|
|
|
1283
1456
|
return '\n'.join(lines) + '\n'
|
|
1284
1457
|
|
|
1285
1458
|
|
|
1459
|
+
# ── Glossary side-channel ─────────────────────────────────────────────────────
|
|
1460
|
+
|
|
1461
|
+
def _relative_source() -> str:
|
|
1462
|
+
vocab_md = SECTIONS_DIR / '01-vocabulary.md'
|
|
1463
|
+
try:
|
|
1464
|
+
return str(vocab_md.relative_to(PROJECT_ROOT))
|
|
1465
|
+
except ValueError:
|
|
1466
|
+
return str(vocab_md)
|
|
1467
|
+
|
|
1468
|
+
|
|
1469
|
+
def write_glossary(vocab: list[dict], stack: dict) -> Path:
|
|
1470
|
+
"""Structured vocabulary for machine consumers.
|
|
1471
|
+
|
|
1472
|
+
01-vocabulary.md stays human-facing. Consumers read this instead of parsing
|
|
1473
|
+
prose: the markdown is a rendered table whose Notes column mixes
|
|
1474
|
+
descriptions with metadata and carries pipe-escaping (`a \\| b`), which is
|
|
1475
|
+
fine to read and hostile to parse.
|
|
1476
|
+
"""
|
|
1477
|
+
entries = [
|
|
1478
|
+
{
|
|
1479
|
+
'key': v.get('alias', ''),
|
|
1480
|
+
'canonical_path': v.get('location', ''),
|
|
1481
|
+
'section': v.get('type', ''),
|
|
1482
|
+
'description': v.get('notes', '') or None,
|
|
1483
|
+
}
|
|
1484
|
+
for v in sorted(vocab, key=lambda x: x.get('alias', ''))
|
|
1485
|
+
if v.get('alias')
|
|
1486
|
+
]
|
|
1487
|
+
payload = {
|
|
1488
|
+
'schema_version': GLOSSARY_SCHEMA_VERSION,
|
|
1489
|
+
'generated_at': datetime.now(timezone.utc).strftime('%Y-%m-%dT%H:%M:%SZ'),
|
|
1490
|
+
# SECTIONS_DIR is bound to the script's location, which is not under
|
|
1491
|
+
# PROJECT_ROOT when --project-root points elsewhere. Report a relative
|
|
1492
|
+
# path when there is one, the absolute path otherwise.
|
|
1493
|
+
'source': _relative_source(),
|
|
1494
|
+
'project': stack.get('name', PROJECT_ROOT.name),
|
|
1495
|
+
'entry_count': len(entries),
|
|
1496
|
+
'entries': entries,
|
|
1497
|
+
}
|
|
1498
|
+
GLOSSARY.write_text(json.dumps(payload, indent=2, ensure_ascii=False) + '\n',
|
|
1499
|
+
encoding='utf-8')
|
|
1500
|
+
return GLOSSARY
|
|
1501
|
+
|
|
1502
|
+
|
|
1286
1503
|
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
1287
1504
|
# ║ PROJECT MAP TOC ║
|
|
1288
1505
|
# ╚══════════════════════════════════════════════════════════════════════════╝
|
|
@@ -1297,6 +1514,7 @@ def build_project_map(
|
|
|
1297
1514
|
services: list[dict],
|
|
1298
1515
|
vocab: list[dict],
|
|
1299
1516
|
section_files: list[tuple[str, Path]],
|
|
1517
|
+
scanned: dict[str, int] | None = None,
|
|
1300
1518
|
) -> str:
|
|
1301
1519
|
now = datetime.now().strftime('%Y-%m-%d %H:%M')
|
|
1302
1520
|
name = stack.get('name', PROJECT_ROOT.name)
|
|
@@ -1315,6 +1533,16 @@ def build_project_map(
|
|
|
1315
1533
|
f"| Docker Services | {len(services)} |",
|
|
1316
1534
|
f"| Vocabulary Entries | {len(vocab)} |",
|
|
1317
1535
|
f"| Stack | {stack.get('language','?')} / {stack.get('framework','?')} |",
|
|
1536
|
+
"",
|
|
1537
|
+
# Inputs beside outputs. "0 routes" alone cannot be judged: from 47
|
|
1538
|
+
# Python files it means an extractor is broken, from 0 it is correct.
|
|
1539
|
+
# grader.py reads these to tell those two cases apart.
|
|
1540
|
+
"### Source files scanned\n",
|
|
1541
|
+
"| Language | Files |",
|
|
1542
|
+
"|----------|-------|",
|
|
1543
|
+
] + [
|
|
1544
|
+
f"| {lang} | {count} |" for lang, count in sorted((scanned or {}).items())
|
|
1545
|
+
] + [
|
|
1318
1546
|
"",
|
|
1319
1547
|
"## Section Index\n",
|
|
1320
1548
|
"| # | Section | Size | When to Read |",
|
|
@@ -1392,7 +1620,7 @@ def main() -> None:
|
|
|
1392
1620
|
|
|
1393
1621
|
# 1. Checksum check
|
|
1394
1622
|
watched = collect_watched_files()
|
|
1395
|
-
checksum = compute_checksum(watched)
|
|
1623
|
+
checksum = compute_checksum(watched, collect_doc_files())
|
|
1396
1624
|
|
|
1397
1625
|
if not args.force and is_unchanged(checksum):
|
|
1398
1626
|
print("[generate] ✓ No changes detected — skipping regeneration (use --force to override)")
|
|
@@ -1434,12 +1662,16 @@ def main() -> None:
|
|
|
1434
1662
|
env_entries = EnvParser().parse()
|
|
1435
1663
|
migrations = MigrationParser().parse()
|
|
1436
1664
|
features = FrontendScanner().scan()
|
|
1665
|
+
skills = SkillParser().parse()
|
|
1437
1666
|
tools = ToolsScanner().scan()
|
|
1438
1667
|
auth_info = AuthScanner().scan()
|
|
1439
1668
|
proxy_rules = ReverseProxyScanner().scan()
|
|
1440
1669
|
|
|
1441
1670
|
print("[generate] Building vocabulary...")
|
|
1442
|
-
vocab = VocabularyBuilder().build(routes, models, schemas, features, stack)
|
|
1671
|
+
vocab = VocabularyBuilder().build(routes, models, schemas, features, stack, skills)
|
|
1672
|
+
|
|
1673
|
+
glossary_path = write_glossary(vocab, stack)
|
|
1674
|
+
print(f"[generate] Glossary → {glossary_path.name} ({len(vocab)} entries)")
|
|
1443
1675
|
|
|
1444
1676
|
print("[generate] Tracing import chains...")
|
|
1445
1677
|
chains = ImportChainTracer().trace(routes)
|
|
@@ -1473,7 +1705,11 @@ def main() -> None:
|
|
|
1473
1705
|
|
|
1474
1706
|
# 6. Write PROJECT_MAP.md
|
|
1475
1707
|
print("[generate] Writing PROJECT_MAP.md...")
|
|
1476
|
-
project_map = build_project_map(
|
|
1708
|
+
project_map = build_project_map(
|
|
1709
|
+
stack, routes, models, schemas, features, migrations, services, vocab, section_files,
|
|
1710
|
+
scanned={'python': len(py_files), 'typescript/javascript': len(ts_files),
|
|
1711
|
+
'go': len(go_files)},
|
|
1712
|
+
)
|
|
1477
1713
|
(MAP_DIR / 'PROJECT_MAP.md').write_text(project_map, encoding='utf-8')
|
|
1478
1714
|
|
|
1479
1715
|
# 7. Update checksums
|
|
@@ -173,7 +173,7 @@ def grade_import_chains() -> GradeResult:
|
|
|
173
173
|
|
|
174
174
|
content = chains_file.read_text(encoding='utf-8', errors='replace')
|
|
175
175
|
|
|
176
|
-
if '_No import chains traced' in content
|
|
176
|
+
if '_No import chains traced' in content:
|
|
177
177
|
# For greenfield or non-Python: acceptable
|
|
178
178
|
details = "No chains traced (greenfield or non-Python stack — ok)"
|
|
179
179
|
return GradeResult('import_chain_validity', WEIGHTS['import_chain_validity'],
|
|
@@ -497,6 +497,17 @@ def print_terminal_summary(results: list[GradeResult], total_score: float, passe
|
|
|
497
497
|
print(f" {'TOTAL SCORE':<30} {score_color}{total_score:5.1f}%{RESET} → {verdict}")
|
|
498
498
|
print(f"{CYAN}{'─' * 55}{RESET}\n")
|
|
499
499
|
|
|
500
|
+
pop, total_sections = populated_sections()
|
|
501
|
+
if total_sections:
|
|
502
|
+
print(f" {'Sections populated':<30} {pop}/{total_sections}"
|
|
503
|
+
f" (diagnostic — not scored)")
|
|
504
|
+
print()
|
|
505
|
+
|
|
506
|
+
for w in usefulness_warnings():
|
|
507
|
+
print(f"{YELLOW} ⚠ {w}{RESET}")
|
|
508
|
+
if usefulness_warnings():
|
|
509
|
+
print()
|
|
510
|
+
|
|
500
511
|
all_issues = [(r.display_name, i) for r in results for i in r.issues]
|
|
501
512
|
if all_issues:
|
|
502
513
|
print(f"{YELLOW} Issues:{RESET}")
|
|
@@ -507,6 +518,71 @@ def print_terminal_summary(results: list[GradeResult], total_score: float, passe
|
|
|
507
518
|
print()
|
|
508
519
|
|
|
509
520
|
|
|
521
|
+
# ── Usefulness diagnostics (reported, never scored) ──────────────────────────
|
|
522
|
+
# The seven graded categories all measure FORM, and generate.py always emits
|
|
523
|
+
# well-formed output — so an empty map and a populated one score identically
|
|
524
|
+
# (both 97.0% before this was added). The information that separates them is
|
|
525
|
+
# not in the map, so it is surfaced as a warning rather than folded into the
|
|
526
|
+
# score: making it scored would fail legitimately sparse repos, which is a
|
|
527
|
+
# worse failure than the one it fixes.
|
|
528
|
+
|
|
529
|
+
def _stat(content: str, label: str) -> int | None:
|
|
530
|
+
m = re.search(rf'^\|\s*{re.escape(label)}\s*\|\s*(\d+)\s*\|', content, re.MULTILINE)
|
|
531
|
+
return int(m.group(1)) if m else None
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
def usefulness_warnings() -> list[str]:
|
|
535
|
+
map_file = MAP_DIR / 'PROJECT_MAP.md'
|
|
536
|
+
if not map_file.exists():
|
|
537
|
+
return []
|
|
538
|
+
content = map_file.read_text(encoding='utf-8', errors='replace')
|
|
539
|
+
warnings = []
|
|
540
|
+
|
|
541
|
+
# Vocabulary is the product. Zero entries means the map gave the user
|
|
542
|
+
# nothing, whatever the seven form categories say. This is the signal that
|
|
543
|
+
# would have surfaced issue #6 at install time.
|
|
544
|
+
vocab = _stat(content, 'Vocabulary Entries')
|
|
545
|
+
if vocab == 0:
|
|
546
|
+
warnings.append(
|
|
547
|
+
"Vocabulary is empty — the map's primary output produced nothing. "
|
|
548
|
+
"Expected on a repo with no source code; otherwise an extractor "
|
|
549
|
+
"does not understand this project's shape."
|
|
550
|
+
)
|
|
551
|
+
|
|
552
|
+
# Source files present but nothing extracted from them: a parser that does
|
|
553
|
+
# not fit the stack, rather than a repo with nothing to find.
|
|
554
|
+
scanned = 0
|
|
555
|
+
block = re.search(r'### Source files scanned(.*?)(?:\n## |\Z)', content, re.DOTALL)
|
|
556
|
+
if block:
|
|
557
|
+
scanned = sum(int(n) for n in re.findall(r'^\|[^|]+\|\s*(\d+)\s*\|',
|
|
558
|
+
block.group(1), re.MULTILINE))
|
|
559
|
+
extracted = sum(v or 0 for v in (
|
|
560
|
+
_stat(content, 'API Routes'), _stat(content, 'Data Models'),
|
|
561
|
+
_stat(content, 'Schemas/DTOs'), _stat(content, 'Frontend Features')))
|
|
562
|
+
if scanned >= 10 and extracted == 0:
|
|
563
|
+
warnings.append(
|
|
564
|
+
f"Scanned {scanned} source file(s) but extracted no routes, models, "
|
|
565
|
+
f"schemas or features — likely an unsupported framework."
|
|
566
|
+
)
|
|
567
|
+
return warnings
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def populated_sections() -> tuple[int, int]:
|
|
571
|
+
"""Sections carrying real content. Diagnostic only — 19 is aspirational,
|
|
572
|
+
not a target: a plugin repo can never populate routes or migrations."""
|
|
573
|
+
files = sorted(SECTIONS_DIR.glob('*.md')) if SECTIONS_DIR.exists() else []
|
|
574
|
+
pop = 0
|
|
575
|
+
for f in files:
|
|
576
|
+
body = f.read_text(encoding='utf-8', errors='replace')
|
|
577
|
+
meat = [ln for ln in body.splitlines()
|
|
578
|
+
if ln.strip() and not ln.startswith(('#', '>'))
|
|
579
|
+
and not re.fullmatch(r'\|[\s|:-]*\|', ln.strip())
|
|
580
|
+
and not re.match(r'^_\(?(no|none|not)\b', ln.strip(), re.I)]
|
|
581
|
+
if meat:
|
|
582
|
+
pop += 1
|
|
583
|
+
return pop, len(files)
|
|
584
|
+
|
|
585
|
+
|
|
510
586
|
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
511
587
|
# ║ MAIN ║
|
|
512
588
|
# ╚══════════════════════════════════════════════════════════════════════════╝
|