@theglitchking/babel-fish 1.0.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/install.sh +666 -0
- package/.claude/project-map/PROJECT_MAP.md +61 -0
- package/.claude/project-map/checksums.json +6 -0
- package/.claude/project-map/generate.py +1481 -0
- package/.claude/project-map/grader.py +583 -0
- package/.claude/project-map/learned-vocabulary.json +1 -0
- package/.claude/project-map/mine-sessions.py +419 -0
- package/.claude/project-map/reports/install-report.md +54 -0
- package/.claude/project-map/reports/iteration-01-report.md +54 -0
- package/.claude/project-map/reports/iteration-01-score.json +7 -0
- package/.claude/project-map/sections/01-vocabulary.md +8 -0
- package/.claude/project-map/sections/02-service-topology.md +6 -0
- package/.claude/project-map/sections/03-environment.md +6 -0
- package/.claude/project-map/sections/04-api-routes.md +6 -0
- package/.claude/project-map/sections/05-data-models.md +4 -0
- package/.claude/project-map/sections/06-schemas.md +4 -0
- package/.claude/project-map/sections/07-services.md +6 -0
- package/.claude/project-map/sections/08-background-jobs.md +5 -0
- package/.claude/project-map/sections/09-frontend-features.md +4 -0
- package/.claude/project-map/sections/10-tools-commands.md +8 -0
- package/.claude/project-map/sections/11-migrations.md +4 -0
- package/.claude/project-map/sections/12-import-chains.md +7 -0
- package/.claude/project-map/sections/13-frontend-backend-map.md +8 -0
- package/.claude/project-map/sections/14-reverse-proxy.md +4 -0
- package/.claude/project-map/sections/15-auth-config.md +6 -0
- package/.claude/project-map/sections/16-infra-profile.md +13 -0
- package/.claude/project-map/sections/17-learned-vocabulary.md +7 -0
- package/.claude/project-map/sections/18-dead-code.md +9 -0
- package/.claude/project-map/sections/19-doc-pointers.md +5 -0
- package/.claude/project-map/stack.json +12 -0
- package/.claude/rules/operational-runbook.md +40 -0
- package/.claude/rules/project-vocabulary.md +25 -0
- package/.claude/scripts/detect-stack.sh +222 -0
- package/.claude/scripts/ensure-python.sh +100 -0
- package/.claude/scripts/statusline.sh +27 -0
- package/.claude/scripts/validate.sh +59 -0
- package/.claude/settings.json +6 -0
- package/.claude/skills/babel-fish-developer-skill/SKILL.md +56 -0
- package/.claude/templates/SKILL.md.template +56 -0
- package/.claude/templates/operational-runbook.md.template +40 -0
- package/.claude/templates/project-vocabulary.md.template +25 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/.claude-plugin/plugin.json +1 -1
- package/.githooks/install.sh +4 -0
- package/.githooks/pre-commit +22 -0
- package/CHANGELOG.md +72 -0
- package/README.md +18 -0
- package/bin/babel-fish.js +78 -71
- package/commands/policy.md +16 -0
- package/commands/relink.md +6 -0
- package/commands/status.md +6 -0
- package/commands/update.md +6 -0
- package/hooks/hooks.json +15 -0
- package/hooks/session-start.js +11 -0
- package/package.json +17 -4
- package/scripts/link-skills.js +31 -0
|
@@ -0,0 +1,1481 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
generate.py — Babel Fish
|
|
4
|
+
Stack-agnostic introspection script for the babel-fish codebase mapper plugin.
|
|
5
|
+
Produces a split-section project map under .claude/project-map/sections/
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
python generate.py [--force] [--project-root PATH] [--stack-json PATH]
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import ast
|
|
13
|
+
import argparse
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
import subprocess
|
|
19
|
+
import sys
|
|
20
|
+
from datetime import datetime
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any
|
|
23
|
+
|
|
24
|
+
# ── Try optional deps ────────────────────────────────────────────────────────
|
|
25
|
+
try:
|
|
26
|
+
import yaml
|
|
27
|
+
HAS_YAML = True
|
|
28
|
+
except ImportError:
|
|
29
|
+
HAS_YAML = False
|
|
30
|
+
|
|
31
|
+
# ── Paths ────────────────────────────────────────────────────────────────────
|
|
32
|
+
SCRIPT_DIR = Path(__file__).parent
|
|
33
|
+
MAP_DIR = SCRIPT_DIR
|
|
34
|
+
SECTIONS_DIR = MAP_DIR / "sections"
|
|
35
|
+
CHECKSUMS = MAP_DIR / "checksums.json"
|
|
36
|
+
LEARNED_VOC = MAP_DIR / "learned-vocabulary.json"
|
|
37
|
+
|
|
38
|
+
# Resolve project root: two levels up from .claude/project-map/
|
|
39
|
+
PROJECT_ROOT = MAP_DIR.parent.parent
|
|
40
|
+
|
|
41
|
+
SECTIONS_DIR.mkdir(parents=True, exist_ok=True)
|
|
42
|
+
|
|
43
|
+
# ── Secrets guard ────────────────────────────────────────────────────────────
|
|
44
|
+
SECRET_PATTERNS = re.compile(
|
|
45
|
+
r'(?i)(password|secret|token|api_key|apikey|private_key|auth_token|'
|
|
46
|
+
r'access_key|secret_key|client_secret|db_pass|database_password|'
|
|
47
|
+
r'stripe_key|twilio_auth|sendgrid_key|aws_secret)\s*[=:]\s*\S+'
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
def redact_secrets(text: str) -> str:
|
|
51
|
+
return SECRET_PATTERNS.sub(r'[REDACTED]', text)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# ── Checksum logic ───────────────────────────────────────────────────────────
|
|
55
|
+
WATCHED_EXTENSIONS = {
|
|
56
|
+
'.py', '.ts', '.tsx', '.js', '.jsx', '.go', '.java', '.kt',
|
|
57
|
+
'.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env',
|
|
58
|
+
}
|
|
59
|
+
WATCHED_NAMES = {
|
|
60
|
+
'docker-compose.yml', 'docker-compose.yaml', 'docker-compose.dev.yml',
|
|
61
|
+
'package.json', 'requirements.txt', 'pyproject.toml', 'go.mod',
|
|
62
|
+
'Cargo.toml', 'pom.xml', 'Gemfile', 'Makefile',
|
|
63
|
+
}
|
|
64
|
+
IGNORE_DIRS = {
|
|
65
|
+
'.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
|
|
66
|
+
'dist', 'build', '.next', '.nuxt', 'target', 'vendor', '.cache',
|
|
67
|
+
'.claude', '.planning',
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
def collect_watched_files() -> list[Path]:
|
|
71
|
+
result = []
|
|
72
|
+
for path in sorted(PROJECT_ROOT.rglob('*')):
|
|
73
|
+
if any(p in IGNORE_DIRS for p in path.parts):
|
|
74
|
+
continue
|
|
75
|
+
if path.is_file() and (path.suffix in WATCHED_EXTENSIONS or path.name in WATCHED_NAMES):
|
|
76
|
+
result.append(path)
|
|
77
|
+
return result
|
|
78
|
+
|
|
79
|
+
def compute_checksum(files: list[Path]) -> str:
|
|
80
|
+
h = hashlib.sha256()
|
|
81
|
+
for f in files:
|
|
82
|
+
h.update(str(f).encode())
|
|
83
|
+
try:
|
|
84
|
+
h.update(str(f.stat().st_mtime_ns).encode())
|
|
85
|
+
except OSError:
|
|
86
|
+
pass
|
|
87
|
+
return h.hexdigest()
|
|
88
|
+
|
|
89
|
+
def load_checksums() -> dict:
|
|
90
|
+
if CHECKSUMS.exists():
|
|
91
|
+
try:
|
|
92
|
+
return json.loads(CHECKSUMS.read_text())
|
|
93
|
+
except Exception:
|
|
94
|
+
pass
|
|
95
|
+
return {}
|
|
96
|
+
|
|
97
|
+
def save_checksums(data: dict) -> None:
|
|
98
|
+
CHECKSUMS.write_text(json.dumps(data, indent=2))
|
|
99
|
+
|
|
100
|
+
def is_unchanged(checksum: str) -> bool:
|
|
101
|
+
stored = load_checksums()
|
|
102
|
+
return stored.get('input_hash') == checksum
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# ── Stack detection (reads stack.json if present, else fallback) ─────────────
|
|
106
|
+
def load_stack(stack_json_path: Path | None = None) -> dict:
|
|
107
|
+
candidates = [
|
|
108
|
+
stack_json_path,
|
|
109
|
+
MAP_DIR / "stack.json",
|
|
110
|
+
PROJECT_ROOT / ".claude" / "stack.json",
|
|
111
|
+
]
|
|
112
|
+
for path in candidates:
|
|
113
|
+
if path and path.exists():
|
|
114
|
+
try:
|
|
115
|
+
return json.loads(path.read_text())
|
|
116
|
+
except Exception:
|
|
117
|
+
pass
|
|
118
|
+
return {
|
|
119
|
+
"name": PROJECT_ROOT.name,
|
|
120
|
+
"slug": re.sub(r'[^a-z0-9]', '-', PROJECT_ROOT.name.lower()).strip('-'),
|
|
121
|
+
"language": "unknown",
|
|
122
|
+
"framework": "unknown",
|
|
123
|
+
"database": "unknown",
|
|
124
|
+
"orm": "unknown",
|
|
125
|
+
"auth": "unknown",
|
|
126
|
+
"package_manager": "unknown",
|
|
127
|
+
"infrastructure": "none",
|
|
128
|
+
"project_root": str(PROJECT_ROOT),
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
133
|
+
# ║ PARSERS ║
|
|
134
|
+
# ╚══════════════════════════════════════════════════════════════════════════╝
|
|
135
|
+
|
|
136
|
+
# ── Python / FastAPI / Django / Flask ────────────────────────────────────────
|
|
137
|
+
|
|
138
|
+
class PythonRouteParser:
|
|
139
|
+
"""Extract routes from Python routers using ast."""
|
|
140
|
+
|
|
141
|
+
DECORATOR_PATTERNS = re.compile(
|
|
142
|
+
r'@(router|app|api_router|blueprint)\.(get|post|put|patch|delete|head|options|websocket)\s*\('
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
146
|
+
routes = []
|
|
147
|
+
for f in files:
|
|
148
|
+
if f.suffix != '.py':
|
|
149
|
+
continue
|
|
150
|
+
try:
|
|
151
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
152
|
+
tree = ast.parse(source, filename=str(f))
|
|
153
|
+
routes.extend(self._extract_routes(tree, f, source))
|
|
154
|
+
except SyntaxError:
|
|
155
|
+
pass
|
|
156
|
+
return routes
|
|
157
|
+
|
|
158
|
+
def _extract_routes(self, tree: ast.AST, filepath: Path, source: str) -> list[dict]:
|
|
159
|
+
routes = []
|
|
160
|
+
lines = source.splitlines()
|
|
161
|
+
for node in ast.walk(tree):
|
|
162
|
+
if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
163
|
+
continue
|
|
164
|
+
for decorator in node.decorator_list:
|
|
165
|
+
info = self._parse_decorator(decorator, lines, node.lineno)
|
|
166
|
+
if info:
|
|
167
|
+
info['file'] = str(filepath.relative_to(PROJECT_ROOT))
|
|
168
|
+
info['line'] = node.lineno
|
|
169
|
+
info['function'] = node.name
|
|
170
|
+
routes.append(info)
|
|
171
|
+
return routes
|
|
172
|
+
|
|
173
|
+
def _parse_decorator(self, decorator: ast.expr, lines: list[str], lineno: int) -> dict | None:
|
|
174
|
+
if isinstance(decorator, ast.Call):
|
|
175
|
+
func = decorator.func
|
|
176
|
+
if isinstance(func, ast.Attribute) and func.attr in (
|
|
177
|
+
'get','post','put','patch','delete','head','options','websocket'
|
|
178
|
+
):
|
|
179
|
+
method = func.attr.upper()
|
|
180
|
+
path = ''
|
|
181
|
+
if decorator.args:
|
|
182
|
+
arg = decorator.args[0]
|
|
183
|
+
if isinstance(arg, ast.Constant):
|
|
184
|
+
path = str(arg.value)
|
|
185
|
+
return {'method': method, 'path': path}
|
|
186
|
+
return None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
class PythonModelParser:
|
|
190
|
+
"""Extract SQLAlchemy / Django ORM models using ast."""
|
|
191
|
+
|
|
192
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
193
|
+
models = []
|
|
194
|
+
for f in files:
|
|
195
|
+
if f.suffix != '.py':
|
|
196
|
+
continue
|
|
197
|
+
try:
|
|
198
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
199
|
+
tree = ast.parse(source, filename=str(f))
|
|
200
|
+
models.extend(self._extract_models(tree, f))
|
|
201
|
+
except SyntaxError:
|
|
202
|
+
pass
|
|
203
|
+
return models
|
|
204
|
+
|
|
205
|
+
def _extract_models(self, tree: ast.AST, filepath: Path) -> list[dict]:
|
|
206
|
+
models = []
|
|
207
|
+
for node in ast.walk(tree):
|
|
208
|
+
if not isinstance(node, ast.ClassDef):
|
|
209
|
+
continue
|
|
210
|
+
bases = [self._base_name(b) for b in node.bases]
|
|
211
|
+
is_model = any(
|
|
212
|
+
b in ('Base', 'Model', 'BaseModel', 'DeclarativeBase', 'AbstractModel')
|
|
213
|
+
or 'Model' in b
|
|
214
|
+
for b in bases if b
|
|
215
|
+
)
|
|
216
|
+
if not is_model:
|
|
217
|
+
continue
|
|
218
|
+
columns = self._extract_columns(node)
|
|
219
|
+
models.append({
|
|
220
|
+
'name': node.name,
|
|
221
|
+
'file': str(filepath.relative_to(PROJECT_ROOT)),
|
|
222
|
+
'line': node.lineno,
|
|
223
|
+
'bases': bases,
|
|
224
|
+
'columns': columns,
|
|
225
|
+
})
|
|
226
|
+
return models
|
|
227
|
+
|
|
228
|
+
def _base_name(self, node: ast.expr) -> str:
|
|
229
|
+
if isinstance(node, ast.Name):
|
|
230
|
+
return node.id
|
|
231
|
+
if isinstance(node, ast.Attribute):
|
|
232
|
+
return node.attr
|
|
233
|
+
return ''
|
|
234
|
+
|
|
235
|
+
def _extract_columns(self, class_node: ast.ClassDef) -> list[dict]:
|
|
236
|
+
cols = []
|
|
237
|
+
for node in ast.walk(class_node):
|
|
238
|
+
if isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name):
|
|
239
|
+
cols.append({
|
|
240
|
+
'name': node.target.id,
|
|
241
|
+
'type': ast.unparse(node.annotation) if hasattr(ast, 'unparse') else '?',
|
|
242
|
+
})
|
|
243
|
+
elif isinstance(node, ast.Assign):
|
|
244
|
+
for target in node.targets:
|
|
245
|
+
if isinstance(target, ast.Name):
|
|
246
|
+
if isinstance(node.value, ast.Call):
|
|
247
|
+
func_name = ''
|
|
248
|
+
if isinstance(node.value.func, ast.Name):
|
|
249
|
+
func_name = node.value.func.id
|
|
250
|
+
elif isinstance(node.value.func, ast.Attribute):
|
|
251
|
+
func_name = node.value.func.attr
|
|
252
|
+
if func_name in ('Column', 'Field', 'CharField', 'IntegerField',
|
|
253
|
+
'TextField', 'BooleanField', 'ForeignKey',
|
|
254
|
+
'ManyToManyField', 'DateTimeField', 'mapped_column'):
|
|
255
|
+
cols.append({'name': target.id, 'type': func_name})
|
|
256
|
+
return cols
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
class PythonSchemaParser:
|
|
260
|
+
"""Extract Pydantic schemas / Django serializers using ast."""
|
|
261
|
+
|
|
262
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
263
|
+
schemas = []
|
|
264
|
+
for f in files:
|
|
265
|
+
if f.suffix != '.py':
|
|
266
|
+
continue
|
|
267
|
+
try:
|
|
268
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
269
|
+
tree = ast.parse(source, filename=str(f))
|
|
270
|
+
schemas.extend(self._extract_schemas(tree, f))
|
|
271
|
+
except SyntaxError:
|
|
272
|
+
pass
|
|
273
|
+
return schemas
|
|
274
|
+
|
|
275
|
+
def _extract_schemas(self, tree: ast.AST, filepath: Path) -> list[dict]:
|
|
276
|
+
schemas = []
|
|
277
|
+
for node in ast.walk(tree):
|
|
278
|
+
if not isinstance(node, ast.ClassDef):
|
|
279
|
+
continue
|
|
280
|
+
bases = [self._base_name(b) for b in node.bases]
|
|
281
|
+
is_schema = any(
|
|
282
|
+
b in ('BaseModel', 'Schema', 'Serializer', 'ModelSerializer',
|
|
283
|
+
'TypedDict', 'NamedTuple')
|
|
284
|
+
or 'Schema' in b or 'Serializer' in b
|
|
285
|
+
for b in bases if b
|
|
286
|
+
)
|
|
287
|
+
if not is_schema:
|
|
288
|
+
continue
|
|
289
|
+
fields = [
|
|
290
|
+
n.target.id
|
|
291
|
+
for n in ast.walk(node)
|
|
292
|
+
if isinstance(n, ast.AnnAssign) and isinstance(n.target, ast.Name)
|
|
293
|
+
]
|
|
294
|
+
schemas.append({
|
|
295
|
+
'name': node.name,
|
|
296
|
+
'file': str(filepath.relative_to(PROJECT_ROOT)),
|
|
297
|
+
'line': node.lineno,
|
|
298
|
+
'fields': fields,
|
|
299
|
+
})
|
|
300
|
+
return schemas
|
|
301
|
+
|
|
302
|
+
def _base_name(self, node: ast.expr) -> str:
|
|
303
|
+
if isinstance(node, ast.Name):
|
|
304
|
+
return node.id
|
|
305
|
+
if isinstance(node, ast.Attribute):
|
|
306
|
+
return node.attr
|
|
307
|
+
return ''
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
# ── TypeScript / JavaScript ──────────────────────────────────────────────────
|
|
311
|
+
|
|
312
|
+
class TSRouteParser:
|
|
313
|
+
"""Extract routes from Next.js, Express, NestJS via regex."""
|
|
314
|
+
|
|
315
|
+
NEXT_APP_ROUTE = re.compile(r'export\s+async\s+function\s+(GET|POST|PUT|PATCH|DELETE|HEAD)\s*\(')
|
|
316
|
+
NEXT_PAGES_ROUTE = re.compile(r'export\s+default\s+(?:async\s+)?function\s+handler')
|
|
317
|
+
EXPRESS_ROUTE = re.compile(r'(?:router|app)\.(get|post|put|patch|delete)\s*\(\s*[\'"]([^\'"]+)[\'"]')
|
|
318
|
+
NEST_DECORATOR = re.compile(r'@(Get|Post|Put|Patch|Delete)\s*\(\s*(?:[\'"]([^\'"]*)[\'"])?\s*\)')
|
|
319
|
+
|
|
320
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
321
|
+
routes = []
|
|
322
|
+
for f in files:
|
|
323
|
+
if f.suffix not in ('.ts', '.tsx', '.js', '.jsx'):
|
|
324
|
+
continue
|
|
325
|
+
try:
|
|
326
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
327
|
+
except Exception:
|
|
328
|
+
continue
|
|
329
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
330
|
+
|
|
331
|
+
# Next.js App Router
|
|
332
|
+
for m in self.NEXT_APP_ROUTE.finditer(source):
|
|
333
|
+
line = source[:m.start()].count('\n') + 1
|
|
334
|
+
path = self._infer_next_path(f)
|
|
335
|
+
routes.append({'method': m.group(1), 'path': path, 'file': rel, 'line': line})
|
|
336
|
+
|
|
337
|
+
# Express
|
|
338
|
+
for m in self.EXPRESS_ROUTE.finditer(source):
|
|
339
|
+
line = source[:m.start()].count('\n') + 1
|
|
340
|
+
routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
|
|
341
|
+
|
|
342
|
+
# NestJS
|
|
343
|
+
for m in self.NEST_DECORATOR.finditer(source):
|
|
344
|
+
line = source[:m.start()].count('\n') + 1
|
|
345
|
+
routes.append({'method': m.group(1).upper(), 'path': m.group(2) or '/', 'file': rel, 'line': line})
|
|
346
|
+
|
|
347
|
+
return routes
|
|
348
|
+
|
|
349
|
+
def _infer_next_path(self, f: Path) -> str:
|
|
350
|
+
"""Infer URL path from Next.js file location."""
|
|
351
|
+
parts = f.parts
|
|
352
|
+
try:
|
|
353
|
+
# app router: after 'app/'
|
|
354
|
+
app_idx = parts.index('app') if 'app' in parts else None
|
|
355
|
+
if app_idx:
|
|
356
|
+
seg = parts[app_idx+1:]
|
|
357
|
+
seg = [s for s in seg if not s.startswith('(') and s not in ('route.ts','route.js','page.tsx','page.ts')]
|
|
358
|
+
return '/' + '/'.join(seg) if seg else '/'
|
|
359
|
+
# pages router
|
|
360
|
+
pages_idx = parts.index('pages') if 'pages' in parts else None
|
|
361
|
+
if pages_idx:
|
|
362
|
+
seg = list(parts[pages_idx+1:])
|
|
363
|
+
seg[-1] = re.sub(r'\.(ts|tsx|js|jsx)$', '', seg[-1])
|
|
364
|
+
if seg[-1] in ('index',):
|
|
365
|
+
seg = seg[:-1]
|
|
366
|
+
return '/' + '/'.join(seg) if seg else '/'
|
|
367
|
+
except (ValueError, IndexError):
|
|
368
|
+
pass
|
|
369
|
+
return f'/{f.stem}'
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
class TSModelParser:
|
|
373
|
+
"""Extract Prisma schema entities and TypeORM entities via regex."""
|
|
374
|
+
|
|
375
|
+
PRISMA_MODEL = re.compile(r'^model\s+(\w+)\s*\{([^}]+)\}', re.MULTILINE)
|
|
376
|
+
PRISMA_FIELD = re.compile(r'^\s+(\w+)\s+(\w+)', re.MULTILINE)
|
|
377
|
+
TYPEORM_ENTITY = re.compile(r'@Entity\s*\(')
|
|
378
|
+
CLASS_NAME = re.compile(r'class\s+(\w+)')
|
|
379
|
+
|
|
380
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
381
|
+
models = []
|
|
382
|
+
for f in files:
|
|
383
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
384
|
+
# Prisma
|
|
385
|
+
if f.suffix == '.prisma':
|
|
386
|
+
try:
|
|
387
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
388
|
+
for m in self.PRISMA_MODEL.finditer(source):
|
|
389
|
+
fields = [
|
|
390
|
+
{'name': fm.group(1), 'type': fm.group(2)}
|
|
391
|
+
for fm in self.PRISMA_FIELD.finditer(m.group(2))
|
|
392
|
+
]
|
|
393
|
+
models.append({'name': m.group(1), 'file': rel, 'columns': fields})
|
|
394
|
+
except Exception:
|
|
395
|
+
pass
|
|
396
|
+
# TypeORM / NestJS entities
|
|
397
|
+
elif f.suffix in ('.ts', '.tsx'):
|
|
398
|
+
try:
|
|
399
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
400
|
+
if self.TYPEORM_ENTITY.search(source):
|
|
401
|
+
cm = self.CLASS_NAME.search(source)
|
|
402
|
+
if cm:
|
|
403
|
+
models.append({'name': cm.group(1), 'file': rel, 'columns': []})
|
|
404
|
+
except Exception:
|
|
405
|
+
pass
|
|
406
|
+
return models
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
# ── Go ───────────────────────────────────────────────────────────────────────
|
|
410
|
+
|
|
411
|
+
class GoRouteParser:
|
|
412
|
+
GIN = re.compile(r'(?:router|r|v\d+|api)\.(GET|POST|PUT|PATCH|DELETE)\s*\(\s*"([^"]+)"')
|
|
413
|
+
CHI = re.compile(r'r\.(Get|Post|Put|Patch|Delete)\s*\(\s*"([^"]+)"')
|
|
414
|
+
STD = re.compile(r'(?:mux|http)\.HandleFunc\s*\(\s*"([^"]+)"')
|
|
415
|
+
|
|
416
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
417
|
+
routes = []
|
|
418
|
+
for f in files:
|
|
419
|
+
if f.suffix != '.go':
|
|
420
|
+
continue
|
|
421
|
+
try:
|
|
422
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
423
|
+
except Exception:
|
|
424
|
+
continue
|
|
425
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
426
|
+
for m in self.GIN.finditer(source):
|
|
427
|
+
line = source[:m.start()].count('\n') + 1
|
|
428
|
+
routes.append({'method': m.group(1), 'path': m.group(2), 'file': rel, 'line': line})
|
|
429
|
+
for m in self.CHI.finditer(source):
|
|
430
|
+
line = source[:m.start()].count('\n') + 1
|
|
431
|
+
routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
|
|
432
|
+
for m in self.STD.finditer(source):
|
|
433
|
+
line = source[:m.start()].count('\n') + 1
|
|
434
|
+
routes.append({'method': 'ANY', 'path': m.group(1), 'file': rel, 'line': line})
|
|
435
|
+
return routes
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
class GoModelParser:
|
|
439
|
+
STRUCT = re.compile(r'type\s+(\w+)\s+struct\s*\{([^}]+)\}', re.DOTALL)
|
|
440
|
+
FIELD = re.compile(r'^\s+(\w+)\s+(\S+)', re.MULTILINE)
|
|
441
|
+
|
|
442
|
+
def parse(self, files: list[Path]) -> list[dict]:
|
|
443
|
+
models = []
|
|
444
|
+
for f in files:
|
|
445
|
+
if f.suffix != '.go':
|
|
446
|
+
continue
|
|
447
|
+
try:
|
|
448
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
449
|
+
except Exception:
|
|
450
|
+
continue
|
|
451
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
452
|
+
for m in self.STRUCT.finditer(source):
|
|
453
|
+
# Only include structs that look like data models (have db/json tags)
|
|
454
|
+
body = m.group(2)
|
|
455
|
+
if 'db:' in body or 'json:' in body or 'gorm:' in body:
|
|
456
|
+
fields = [
|
|
457
|
+
{'name': fm.group(1), 'type': fm.group(2)}
|
|
458
|
+
for fm in self.FIELD.finditer(body)
|
|
459
|
+
if not fm.group(1).startswith('//')
|
|
460
|
+
]
|
|
461
|
+
models.append({'name': m.group(1), 'file': rel, 'columns': fields})
|
|
462
|
+
return models
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
# ── Docker Compose ───────────────────────────────────────────────────────────
|
|
466
|
+
|
|
467
|
+
class DockerComposeParser:
|
|
468
|
+
def parse(self) -> list[dict]:
|
|
469
|
+
services = []
|
|
470
|
+
candidates = [
|
|
471
|
+
'docker-compose.yml', 'docker-compose.yaml',
|
|
472
|
+
'docker-compose.dev.yml', 'docker-compose.development.yml',
|
|
473
|
+
'docker-compose.prod.yml', 'docker-compose.production.yml',
|
|
474
|
+
]
|
|
475
|
+
for name in candidates:
|
|
476
|
+
path = PROJECT_ROOT / name
|
|
477
|
+
if not path.exists():
|
|
478
|
+
continue
|
|
479
|
+
try:
|
|
480
|
+
data = self._load(path)
|
|
481
|
+
if data and isinstance(data.get('services'), dict):
|
|
482
|
+
for svc_name, svc in data['services'].items():
|
|
483
|
+
services.append({
|
|
484
|
+
'name': svc_name,
|
|
485
|
+
'image': svc.get('image', svc.get('build', '(build)')),
|
|
486
|
+
'ports': svc.get('ports', []),
|
|
487
|
+
'depends_on': svc.get('depends_on', []),
|
|
488
|
+
'environment_keys': list(svc.get('environment', {}).keys())
|
|
489
|
+
if isinstance(svc.get('environment'), dict)
|
|
490
|
+
else [],
|
|
491
|
+
'source_file': name,
|
|
492
|
+
})
|
|
493
|
+
except Exception:
|
|
494
|
+
pass
|
|
495
|
+
return services
|
|
496
|
+
|
|
497
|
+
def _load(self, path: Path) -> dict | None:
|
|
498
|
+
text = path.read_text(encoding='utf-8', errors='replace')
|
|
499
|
+
if HAS_YAML:
|
|
500
|
+
return yaml.safe_load(text)
|
|
501
|
+
# Regex fallback: extract service names at minimum
|
|
502
|
+
services: dict = {'services': {}}
|
|
503
|
+
in_services = False
|
|
504
|
+
indent = 0
|
|
505
|
+
for line in text.splitlines():
|
|
506
|
+
if re.match(r'^services\s*:', line):
|
|
507
|
+
in_services = True
|
|
508
|
+
continue
|
|
509
|
+
if in_services:
|
|
510
|
+
m = re.match(r'^(\s+)(\w[\w-]*)\s*:', line)
|
|
511
|
+
if m:
|
|
512
|
+
lvl = len(m.group(1))
|
|
513
|
+
if indent == 0:
|
|
514
|
+
indent = lvl
|
|
515
|
+
if lvl == indent:
|
|
516
|
+
services['services'][m.group(2)] = {}
|
|
517
|
+
return services
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
# ── Env Parser ───────────────────────────────────────────────────────────────
|
|
521
|
+
|
|
522
|
+
class EnvParser:
|
|
523
|
+
SECRET_KEYS = re.compile(
|
|
524
|
+
r'(?i)(password|secret|token|key|auth|credential|private|cert)'
|
|
525
|
+
)
|
|
526
|
+
|
|
527
|
+
def parse(self) -> list[dict]:
|
|
528
|
+
entries = []
|
|
529
|
+
candidates = ['.env', '.env.example', '.env.local', '.env.development', '.env.sample']
|
|
530
|
+
for name in candidates:
|
|
531
|
+
path = PROJECT_ROOT / name
|
|
532
|
+
if not path.exists():
|
|
533
|
+
continue
|
|
534
|
+
for line in path.read_text(encoding='utf-8', errors='replace').splitlines():
|
|
535
|
+
line = line.strip()
|
|
536
|
+
if not line or line.startswith('#'):
|
|
537
|
+
continue
|
|
538
|
+
if '=' not in line:
|
|
539
|
+
continue
|
|
540
|
+
key, _, raw_val = line.partition('=')
|
|
541
|
+
key = key.strip()
|
|
542
|
+
val = raw_val.strip()
|
|
543
|
+
is_secret = bool(self.SECRET_KEYS.search(key))
|
|
544
|
+
entries.append({
|
|
545
|
+
'key': key,
|
|
546
|
+
'value': '[SECRET]' if is_secret else val[:80],
|
|
547
|
+
'is_secret': is_secret,
|
|
548
|
+
'source': name,
|
|
549
|
+
})
|
|
550
|
+
return entries
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
# ── Migration Parser ─────────────────────────────────────────────────────────
|
|
554
|
+
|
|
555
|
+
class MigrationParser:
|
|
556
|
+
def parse(self) -> list[dict]:
|
|
557
|
+
migrations = []
|
|
558
|
+
# Alembic
|
|
559
|
+
alembic_dirs = list(PROJECT_ROOT.rglob('versions'))
|
|
560
|
+
for d in alembic_dirs:
|
|
561
|
+
if not d.is_dir():
|
|
562
|
+
continue
|
|
563
|
+
for f in sorted(d.glob('*.py')):
|
|
564
|
+
try:
|
|
565
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
566
|
+
rev = re.search(r'revision\s*=\s*[\'"]([^\'"]+)[\'"]', source)
|
|
567
|
+
down = re.search(r'down_revision\s*=\s*[\'"]?([^\'")\n]+)[\'"]?', source)
|
|
568
|
+
msg = re.search(r'Create Date.*\n.*"""(.+?)"""', source, re.DOTALL)
|
|
569
|
+
tables = re.findall(r'op\.(?:create_table|drop_table|add_column)\s*\(\s*[\'"]([^\'"]+)[\'"]', source)
|
|
570
|
+
migrations.append({
|
|
571
|
+
'revision': rev.group(1) if rev else f.stem,
|
|
572
|
+
'down_revision': down.group(1).strip() if down else None,
|
|
573
|
+
'tables_affected': list(set(tables)),
|
|
574
|
+
'file': str(f.relative_to(PROJECT_ROOT)),
|
|
575
|
+
'type': 'alembic',
|
|
576
|
+
})
|
|
577
|
+
except Exception:
|
|
578
|
+
pass
|
|
579
|
+
|
|
580
|
+
# SQL files
|
|
581
|
+
for f in sorted(PROJECT_ROOT.rglob('*.sql')):
|
|
582
|
+
if any(p in IGNORE_DIRS for p in f.parts):
|
|
583
|
+
continue
|
|
584
|
+
try:
|
|
585
|
+
source = f.read_text(encoding='utf-8', errors='replace')[:2000]
|
|
586
|
+
tables = re.findall(r'(?:CREATE|ALTER|DROP)\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?(\w+)', source, re.IGNORECASE)
|
|
587
|
+
if tables:
|
|
588
|
+
migrations.append({
|
|
589
|
+
'revision': f.stem,
|
|
590
|
+
'tables_affected': list(set(tables)),
|
|
591
|
+
'file': str(f.relative_to(PROJECT_ROOT)),
|
|
592
|
+
'type': 'sql',
|
|
593
|
+
})
|
|
594
|
+
except Exception:
|
|
595
|
+
pass
|
|
596
|
+
|
|
597
|
+
return migrations
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
# ── Frontend Feature Scanner ─────────────────────────────────────────────────
|
|
601
|
+
|
|
602
|
+
class FrontendScanner:
|
|
603
|
+
FEATURE_DIRS = {'features', 'pages', 'views', 'screens', 'modules', 'app'}
|
|
604
|
+
COMPONENT_EXTS = {'.tsx', '.jsx'}
|
|
605
|
+
HOOK_PATTERN = re.compile(r'^use[A-Z]')
|
|
606
|
+
|
|
607
|
+
def scan(self) -> list[dict]:
|
|
608
|
+
features = []
|
|
609
|
+
for feat_dir_name in self.FEATURE_DIRS:
|
|
610
|
+
feat_dir = PROJECT_ROOT / feat_dir_name
|
|
611
|
+
if not feat_dir.exists():
|
|
612
|
+
# try nested src/
|
|
613
|
+
feat_dir = PROJECT_ROOT / 'src' / feat_dir_name
|
|
614
|
+
if not feat_dir.is_dir():
|
|
615
|
+
continue
|
|
616
|
+
for entry in sorted(feat_dir.iterdir()):
|
|
617
|
+
if entry.is_dir() and not entry.name.startswith('.'):
|
|
618
|
+
stats = self._scan_feature_dir(entry)
|
|
619
|
+
stats['name'] = entry.name
|
|
620
|
+
stats['path'] = str(entry.relative_to(PROJECT_ROOT))
|
|
621
|
+
features.append(stats)
|
|
622
|
+
return features
|
|
623
|
+
|
|
624
|
+
def _scan_feature_dir(self, d: Path) -> dict:
|
|
625
|
+
components, hooks, stores, api_files = [], [], [], []
|
|
626
|
+
for f in d.rglob('*'):
|
|
627
|
+
if f.suffix in self.COMPONENT_EXTS:
|
|
628
|
+
if self.HOOK_PATTERN.match(f.stem):
|
|
629
|
+
hooks.append(f.name)
|
|
630
|
+
else:
|
|
631
|
+
components.append(f.name)
|
|
632
|
+
elif 'store' in f.name.lower() or 'slice' in f.name.lower():
|
|
633
|
+
stores.append(f.name)
|
|
634
|
+
elif 'api' in f.name.lower() or 'service' in f.name.lower() or 'client' in f.name.lower():
|
|
635
|
+
api_files.append(f.name)
|
|
636
|
+
return {
|
|
637
|
+
'components': components,
|
|
638
|
+
'hooks': hooks,
|
|
639
|
+
'stores': stores,
|
|
640
|
+
'api_files': api_files,
|
|
641
|
+
'component_count': len(components),
|
|
642
|
+
'hook_count': len(hooks),
|
|
643
|
+
}
|
|
644
|
+
|
|
645
|
+
|
|
646
|
+
# ── Import Chain Tracer ───────────────────────────────────────────────────────
|
|
647
|
+
|
|
648
|
+
class ImportChainTracer:
|
|
649
|
+
MAX_DEPTH = 5
|
|
650
|
+
|
|
651
|
+
def trace(self, routes: list[dict]) -> list[dict]:
|
|
652
|
+
chains = []
|
|
653
|
+
for route in routes[:30]: # limit to first 30 routes
|
|
654
|
+
f = PROJECT_ROOT / route.get('file', '')
|
|
655
|
+
if not f.exists() or f.suffix != '.py':
|
|
656
|
+
continue
|
|
657
|
+
chain = self._trace_file(f, depth=0)
|
|
658
|
+
if len(chain) > 1:
|
|
659
|
+
chains.append({
|
|
660
|
+
'route': f"{route.get('method','?')} {route.get('path','?')}",
|
|
661
|
+
'chain': ' → '.join(chain),
|
|
662
|
+
'file': route.get('file'),
|
|
663
|
+
})
|
|
664
|
+
return chains
|
|
665
|
+
|
|
666
|
+
def _trace_file(self, f: Path, depth: int) -> list[str]:
|
|
667
|
+
if depth > self.MAX_DEPTH or not f.exists():
|
|
668
|
+
return []
|
|
669
|
+
result = [str(f.relative_to(PROJECT_ROOT))]
|
|
670
|
+
try:
|
|
671
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
672
|
+
tree = ast.parse(source)
|
|
673
|
+
for node in ast.walk(tree):
|
|
674
|
+
if isinstance(node, (ast.Import, ast.ImportFrom)):
|
|
675
|
+
if isinstance(node, ast.ImportFrom) and node.module:
|
|
676
|
+
# Attempt to resolve local import
|
|
677
|
+
parts = node.module.split('.')
|
|
678
|
+
candidate = PROJECT_ROOT / Path(*parts).with_suffix('.py')
|
|
679
|
+
if candidate.exists() and depth < self.MAX_DEPTH:
|
|
680
|
+
sub = self._trace_file(candidate, depth + 1)
|
|
681
|
+
if sub:
|
|
682
|
+
result.extend(sub[1:])
|
|
683
|
+
break
|
|
684
|
+
except Exception:
|
|
685
|
+
pass
|
|
686
|
+
return result
|
|
687
|
+
|
|
688
|
+
|
|
689
|
+
# ── Vocabulary Builder ────────────────────────────────────────────────────────
|
|
690
|
+
|
|
691
|
+
class VocabularyBuilder:
|
|
692
|
+
SKIP_WORDS = {'index', 'main', 'base', 'common', 'utils', 'helpers', 'types',
|
|
693
|
+
'constants', 'config', 'lib', 'api', 'app', 'src', 'test', 'tests'}
|
|
694
|
+
|
|
695
|
+
def build(
|
|
696
|
+
self,
|
|
697
|
+
routes: list[dict],
|
|
698
|
+
models: list[dict],
|
|
699
|
+
schemas: list[dict],
|
|
700
|
+
features: list[dict],
|
|
701
|
+
stack: dict,
|
|
702
|
+
) -> list[dict]:
|
|
703
|
+
vocab: dict[str, dict] = {}
|
|
704
|
+
|
|
705
|
+
# From features
|
|
706
|
+
for feat in features:
|
|
707
|
+
name = feat['name']
|
|
708
|
+
if name.lower() in self.SKIP_WORDS:
|
|
709
|
+
continue
|
|
710
|
+
aliases = self._name_to_aliases(name)
|
|
711
|
+
for alias in aliases:
|
|
712
|
+
self._add(vocab, alias, 'feature', feat.get('path', ''), f"{feat['component_count']} components")
|
|
713
|
+
|
|
714
|
+
# From models
|
|
715
|
+
for model in models:
|
|
716
|
+
aliases = self._name_to_aliases(model['name'])
|
|
717
|
+
for alias in aliases:
|
|
718
|
+
self._add(vocab, alias, 'model', model.get('file', ''), f"table/model: {model['name']}")
|
|
719
|
+
|
|
720
|
+
# From routes (group by path prefix)
|
|
721
|
+
route_groups: dict[str, list] = {}
|
|
722
|
+
for route in routes:
|
|
723
|
+
prefix = route.get('path', '/').split('/')[1] if '/' in route.get('path', '/') else route.get('path', '')
|
|
724
|
+
if prefix and prefix not in self.SKIP_WORDS:
|
|
725
|
+
route_groups.setdefault(prefix, []).append(route)
|
|
726
|
+
for prefix, group in route_groups.items():
|
|
727
|
+
aliases = self._name_to_aliases(prefix)
|
|
728
|
+
file_ex = group[0].get('file', '')
|
|
729
|
+
for alias in aliases:
|
|
730
|
+
self._add(vocab, alias, 'api', file_ex, f"{len(group)} routes")
|
|
731
|
+
|
|
732
|
+
# Merge learned vocabulary
|
|
733
|
+
learned = self._load_learned()
|
|
734
|
+
for alias, data in learned.items():
|
|
735
|
+
if alias not in vocab and data.get('score', 0) >= 5:
|
|
736
|
+
vocab[alias] = {
|
|
737
|
+
'alias': alias,
|
|
738
|
+
'type': 'learned',
|
|
739
|
+
'location': ', '.join(data.get('targets', [])),
|
|
740
|
+
'notes': f"learned from session (score: {data['score']:.1f})",
|
|
741
|
+
}
|
|
742
|
+
|
|
743
|
+
return list(vocab.values())
|
|
744
|
+
|
|
745
|
+
def _name_to_aliases(self, name: str) -> list[str]:
|
|
746
|
+
"""Generate human aliases from a code name."""
|
|
747
|
+
# camelCase / PascalCase → words
|
|
748
|
+
words = re.sub(r'([A-Z])', r' \1', name).lower().split()
|
|
749
|
+
words = [w for w in words if w and w not in self.SKIP_WORDS]
|
|
750
|
+
if not words:
|
|
751
|
+
return []
|
|
752
|
+
phrase = ' '.join(words)
|
|
753
|
+
aliases = [phrase, name.lower()]
|
|
754
|
+
# e.g. 'deal-pipeline' → 'deal pipeline'
|
|
755
|
+
if '-' in name or '_' in name:
|
|
756
|
+
clean = re.sub(r'[-_]', ' ', name).lower()
|
|
757
|
+
aliases.append(clean)
|
|
758
|
+
return list(dict.fromkeys(aliases)) # dedupe, preserve order
|
|
759
|
+
|
|
760
|
+
def _add(self, vocab: dict, alias: str, type_: str, location: str, notes: str) -> None:
|
|
761
|
+
if alias and alias not in vocab:
|
|
762
|
+
vocab[alias] = {'alias': alias, 'type': type_, 'location': location, 'notes': notes}
|
|
763
|
+
|
|
764
|
+
def _load_learned(self) -> dict:
|
|
765
|
+
if LEARNED_VOC.exists():
|
|
766
|
+
try:
|
|
767
|
+
return json.loads(LEARNED_VOC.read_text())
|
|
768
|
+
except Exception:
|
|
769
|
+
pass
|
|
770
|
+
return {}
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
# ── Tools & Commands Scanner ──────────────────────────────────────────────────
|
|
774
|
+
|
|
775
|
+
class ToolsScanner:
|
|
776
|
+
def scan(self) -> list[dict]:
|
|
777
|
+
tools = []
|
|
778
|
+
|
|
779
|
+
# npm scripts
|
|
780
|
+
pkg = PROJECT_ROOT / 'package.json'
|
|
781
|
+
if pkg.exists():
|
|
782
|
+
try:
|
|
783
|
+
data = json.loads(pkg.read_text())
|
|
784
|
+
for name, cmd in (data.get('scripts') or {}).items():
|
|
785
|
+
tools.append({'name': name, 'command': f'npm run {name}', 'description': cmd, 'source': 'package.json'})
|
|
786
|
+
except Exception:
|
|
787
|
+
pass
|
|
788
|
+
|
|
789
|
+
# Makefile targets
|
|
790
|
+
makefile = PROJECT_ROOT / 'Makefile'
|
|
791
|
+
if makefile.exists():
|
|
792
|
+
for line in makefile.read_text(encoding='utf-8', errors='replace').splitlines():
|
|
793
|
+
m = re.match(r'^([a-zA-Z][a-zA-Z0-9_-]+)\s*:', line)
|
|
794
|
+
if m and not m.group(1).startswith('.'):
|
|
795
|
+
tools.append({'name': m.group(1), 'command': f'make {m.group(1)}', 'description': '', 'source': 'Makefile'})
|
|
796
|
+
|
|
797
|
+
# Poetry / pip scripts
|
|
798
|
+
for cfg in ['pyproject.toml', 'setup.cfg']:
|
|
799
|
+
path = PROJECT_ROOT / cfg
|
|
800
|
+
if path.exists():
|
|
801
|
+
text = path.read_text(encoding='utf-8', errors='replace')
|
|
802
|
+
for m in re.finditer(r'^\[tool\.poetry\.scripts\]\s*\n((?:\w.*\n)*)', text, re.MULTILINE):
|
|
803
|
+
for line in m.group(1).splitlines():
|
|
804
|
+
if '=' in line:
|
|
805
|
+
name = line.split('=')[0].strip()
|
|
806
|
+
tools.append({'name': name, 'command': name, 'description': '', 'source': cfg})
|
|
807
|
+
|
|
808
|
+
# Shell scripts at root
|
|
809
|
+
for f in PROJECT_ROOT.glob('*.sh'):
|
|
810
|
+
tools.append({'name': f.name, 'command': f'bash {f.name}', 'description': '', 'source': 'shell'})
|
|
811
|
+
|
|
812
|
+
# .claude/skills
|
|
813
|
+
skills_dir = PROJECT_ROOT / '.claude' / 'skills'
|
|
814
|
+
if skills_dir.is_dir():
|
|
815
|
+
for skill_dir in skills_dir.iterdir():
|
|
816
|
+
if (skill_dir / 'SKILL.md').exists():
|
|
817
|
+
tools.append({'name': f'/{skill_dir.name}', 'command': f'/{skill_dir.name}', 'description': 'Claude skill', 'source': 'skills'})
|
|
818
|
+
|
|
819
|
+
return tools
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
# ── Auth Config Scanner ───────────────────────────────────────────────────────
|
|
823
|
+
|
|
824
|
+
class AuthScanner:
|
|
825
|
+
def scan(self) -> dict:
|
|
826
|
+
info: dict[str, Any] = {'provider': 'unknown', 'files': [], 'patterns': []}
|
|
827
|
+
|
|
828
|
+
patterns_map = {
|
|
829
|
+
'supabase': ['supabase', 'createClient', 'auth.signIn'],
|
|
830
|
+
'next-auth': ['NextAuth', 'getSession', 'useSession', 'SessionProvider'],
|
|
831
|
+
'clerk': ['ClerkProvider', 'useUser', '@clerk'],
|
|
832
|
+
'auth0': ['Auth0Provider', 'useAuth0', '@auth0'],
|
|
833
|
+
'jwt': ['jwt.sign', 'jwt.verify', 'create_access_token', 'decode_token'],
|
|
834
|
+
'passport': ['passport.use', 'passport.authenticate'],
|
|
835
|
+
'firebase': ['initializeApp', 'getAuth', 'signInWithEmailAndPassword'],
|
|
836
|
+
}
|
|
837
|
+
|
|
838
|
+
for f in PROJECT_ROOT.rglob('*'):
|
|
839
|
+
if any(p in IGNORE_DIRS for p in f.parts):
|
|
840
|
+
continue
|
|
841
|
+
if f.suffix not in ('.py', '.ts', '.tsx', '.js', '.jsx'):
|
|
842
|
+
continue
|
|
843
|
+
try:
|
|
844
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
845
|
+
except Exception:
|
|
846
|
+
continue
|
|
847
|
+
for provider, patterns in patterns_map.items():
|
|
848
|
+
if any(p in source for p in patterns):
|
|
849
|
+
info['provider'] = provider
|
|
850
|
+
rel = str(f.relative_to(PROJECT_ROOT))
|
|
851
|
+
if rel not in info['files']:
|
|
852
|
+
info['files'].append(rel)
|
|
853
|
+
info['patterns'].extend([p for p in patterns if p in source and p not in info['patterns']])
|
|
854
|
+
|
|
855
|
+
return info
|
|
856
|
+
|
|
857
|
+
|
|
858
|
+
# ── Reverse Proxy Scanner ─────────────────────────────────────────────────────
|
|
859
|
+
|
|
860
|
+
class ReverseProxyScanner:
|
|
861
|
+
def scan(self) -> list[dict]:
|
|
862
|
+
rules = []
|
|
863
|
+
# nginx
|
|
864
|
+
for f in PROJECT_ROOT.rglob('*.conf'):
|
|
865
|
+
if any(p in IGNORE_DIRS for p in f.parts):
|
|
866
|
+
continue
|
|
867
|
+
try:
|
|
868
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
869
|
+
for m in re.finditer(r'location\s+([^\s{]+)\s*\{[^}]*proxy_pass\s+([^;]+);', source, re.DOTALL):
|
|
870
|
+
rules.append({'type': 'nginx', 'location': m.group(1).strip(), 'upstream': m.group(2).strip(), 'file': str(f.relative_to(PROJECT_ROOT))})
|
|
871
|
+
except Exception:
|
|
872
|
+
pass
|
|
873
|
+
# Caddyfile
|
|
874
|
+
caddyfile = PROJECT_ROOT / 'Caddyfile'
|
|
875
|
+
if caddyfile.exists():
|
|
876
|
+
try:
|
|
877
|
+
source = caddyfile.read_text(encoding='utf-8', errors='replace')
|
|
878
|
+
for m in re.finditer(r'reverse_proxy\s+([^\n]+)', source):
|
|
879
|
+
rules.append({'type': 'caddy', 'location': '*', 'upstream': m.group(1).strip(), 'file': 'Caddyfile'})
|
|
880
|
+
except Exception:
|
|
881
|
+
pass
|
|
882
|
+
return rules
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
# ── Dead Code Detector ────────────────────────────────────────────────────────
|
|
886
|
+
|
|
887
|
+
class DeadCodeDetector:
|
|
888
|
+
def detect(self, vocab: list[dict]) -> list[dict]:
|
|
889
|
+
candidates = []
|
|
890
|
+
cutoff_date = '--since=6 months ago'
|
|
891
|
+
|
|
892
|
+
low_score_files = set()
|
|
893
|
+
if LEARNED_VOC.exists():
|
|
894
|
+
try:
|
|
895
|
+
learned = json.loads(LEARNED_VOC.read_text())
|
|
896
|
+
for alias, data in learned.items():
|
|
897
|
+
if data.get('score', 999) < 2:
|
|
898
|
+
low_score_files.update(data.get('targets', []))
|
|
899
|
+
except Exception:
|
|
900
|
+
pass
|
|
901
|
+
|
|
902
|
+
for filepath_str in low_score_files:
|
|
903
|
+
path = PROJECT_ROOT / filepath_str
|
|
904
|
+
if not path.exists():
|
|
905
|
+
continue
|
|
906
|
+
try:
|
|
907
|
+
result = subprocess.run(
|
|
908
|
+
['git', '-C', str(PROJECT_ROOT), 'log', cutoff_date, '--', filepath_str],
|
|
909
|
+
capture_output=True, text=True, timeout=5
|
|
910
|
+
)
|
|
911
|
+
if not result.stdout.strip():
|
|
912
|
+
candidates.append({
|
|
913
|
+
'file': filepath_str,
|
|
914
|
+
'reason': 'No commits in 6 months + low vocabulary score',
|
|
915
|
+
})
|
|
916
|
+
except Exception:
|
|
917
|
+
pass
|
|
918
|
+
|
|
919
|
+
return candidates
|
|
920
|
+
|
|
921
|
+
|
|
922
|
+
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
923
|
+
# ║ SECTION WRITERS ║
|
|
924
|
+
# ╚══════════════════════════════════════════════════════════════════════════╝
|
|
925
|
+
|
|
926
|
+
def write_section(num: str, name: str, content: str) -> Path:
|
|
927
|
+
filename = f"{num}-{name}.md"
|
|
928
|
+
path = SECTIONS_DIR / filename
|
|
929
|
+
path.write_text(content, encoding='utf-8')
|
|
930
|
+
return path
|
|
931
|
+
|
|
932
|
+
def section_size_kb(path: Path) -> float:
|
|
933
|
+
return path.stat().st_size / 1024 if path.exists() else 0.0
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def build_vocabulary_section(vocab: list[dict]) -> str:
|
|
937
|
+
lines = [
|
|
938
|
+
"# Section 01 — Vocabulary Translation Layer\n",
|
|
939
|
+
"> Maps human language to exact code locations. Auto-generated.\n\n",
|
|
940
|
+
"| Alias | Type | Location | Notes |",
|
|
941
|
+
"|-------|------|----------|-------|",
|
|
942
|
+
]
|
|
943
|
+
for v in sorted(vocab, key=lambda x: x.get('alias', '')):
|
|
944
|
+
alias = v.get('alias', '').replace('|', '\\|')
|
|
945
|
+
type_ = v.get('type', '').replace('|', '\\|')
|
|
946
|
+
location = v.get('location', '').replace('|', '\\|')
|
|
947
|
+
notes = v.get('notes', '').replace('|', '\\|')
|
|
948
|
+
lines.append(f"| {alias} | {type_} | {location} | {notes} |")
|
|
949
|
+
if not vocab:
|
|
950
|
+
lines.append("| _(no vocabulary generated yet — add source code to populate)_ | | | |")
|
|
951
|
+
return '\n'.join(lines) + '\n'
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
def build_topology_section(services: list[dict]) -> str:
|
|
955
|
+
lines = [
|
|
956
|
+
"# Section 02 — Service Topology\n",
|
|
957
|
+
"> Docker services, ports, and dependencies.\n\n",
|
|
958
|
+
]
|
|
959
|
+
if not services:
|
|
960
|
+
lines.append("_No docker-compose services detected._\n")
|
|
961
|
+
return '\n'.join(lines)
|
|
962
|
+
lines += ["| Service | Image | Ports | Depends On |",
|
|
963
|
+
"|---------|-------|-------|------------|"]
|
|
964
|
+
for svc in services:
|
|
965
|
+
ports = ', '.join(str(p) for p in svc.get('ports', []))
|
|
966
|
+
deps = ', '.join(svc.get('depends_on', []) if isinstance(svc.get('depends_on'), list)
|
|
967
|
+
else list(svc.get('depends_on', {}).keys()))
|
|
968
|
+
img = str(svc.get('image', '?'))[:50]
|
|
969
|
+
lines.append(f"| {svc['name']} | {img} | {ports} | {deps} |")
|
|
970
|
+
return '\n'.join(lines) + '\n'
|
|
971
|
+
|
|
972
|
+
|
|
973
|
+
def build_environment_section(env_entries: list[dict]) -> str:
|
|
974
|
+
lines = [
|
|
975
|
+
"# Section 03 — Environment Variables\n",
|
|
976
|
+
"> Key names and non-sensitive values only. Secrets are redacted.\n\n",
|
|
977
|
+
]
|
|
978
|
+
if not env_entries:
|
|
979
|
+
lines.append("_No .env files found._\n")
|
|
980
|
+
return '\n'.join(lines)
|
|
981
|
+
lines += ["| Key | Value | Source |",
|
|
982
|
+
"|-----|-------|--------|"]
|
|
983
|
+
for e in env_entries:
|
|
984
|
+
val = '[SECRET]' if e.get('is_secret') else str(e.get('value', ''))[:60]
|
|
985
|
+
lines.append(f"| `{e['key']}` | `{val}` | {e.get('source', '')} |")
|
|
986
|
+
return '\n'.join(lines) + '\n'
|
|
987
|
+
|
|
988
|
+
|
|
989
|
+
def build_routes_section(routes: list[dict]) -> str:
|
|
990
|
+
lines = [
|
|
991
|
+
"# Section 04 — API Routes\n\n",
|
|
992
|
+
"| Method | Path | File | Line |",
|
|
993
|
+
"|--------|------|------|------|",
|
|
994
|
+
]
|
|
995
|
+
for r in sorted(routes, key=lambda x: (x.get('path',''), x.get('method',''))):
|
|
996
|
+
lines.append(f"| `{r.get('method','?')}` | `{r.get('path','?')}` | {r.get('file','')} | {r.get('line','')} |")
|
|
997
|
+
if not routes:
|
|
998
|
+
lines.append("| _(no routes detected yet)_ | | | |")
|
|
999
|
+
return '\n'.join(lines) + '\n'
|
|
1000
|
+
|
|
1001
|
+
|
|
1002
|
+
def build_models_section(models: list[dict]) -> str:
|
|
1003
|
+
lines = ["# Section 05 — Data Models\n\n"]
|
|
1004
|
+
if not models:
|
|
1005
|
+
lines.append("_No data models detected yet._\n")
|
|
1006
|
+
return '\n'.join(lines)
|
|
1007
|
+
for model in models:
|
|
1008
|
+
lines.append(f"## {model['name']}")
|
|
1009
|
+
lines.append(f"- **File**: `{model.get('file','?')}`")
|
|
1010
|
+
cols = model.get('columns', [])
|
|
1011
|
+
if cols:
|
|
1012
|
+
lines.append("- **Fields**:")
|
|
1013
|
+
for c in cols[:20]:
|
|
1014
|
+
lines.append(f" - `{c.get('name','?')}` ({c.get('type','?')})")
|
|
1015
|
+
lines.append('')
|
|
1016
|
+
return '\n'.join(lines)
|
|
1017
|
+
|
|
1018
|
+
|
|
1019
|
+
def build_schemas_section(schemas: list[dict]) -> str:
|
|
1020
|
+
lines = ["# Section 06 — Schemas / DTOs\n\n"]
|
|
1021
|
+
if not schemas:
|
|
1022
|
+
lines.append("_No schemas detected yet._\n")
|
|
1023
|
+
return '\n'.join(lines)
|
|
1024
|
+
for s in schemas:
|
|
1025
|
+
lines.append(f"## {s['name']}")
|
|
1026
|
+
lines.append(f"- **File**: `{s.get('file','?')}`")
|
|
1027
|
+
fields = s.get('fields', [])
|
|
1028
|
+
if fields:
|
|
1029
|
+
lines.append(f"- **Fields**: {', '.join(f'`{f}`' for f in fields[:15])}")
|
|
1030
|
+
lines.append('')
|
|
1031
|
+
return '\n'.join(lines)
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
def build_services_section(routes: list[dict], models: list[dict]) -> str:
|
|
1035
|
+
lines = ["# Section 07 — Services\n\n",
|
|
1036
|
+
"_Service discovery is inferred from directory structure and imports._\n\n"]
|
|
1037
|
+
# Collect unique directories containing routes or models
|
|
1038
|
+
dirs: dict[str, int] = {}
|
|
1039
|
+
for item in routes + models:
|
|
1040
|
+
f = item.get('file', '')
|
|
1041
|
+
d = str(Path(f).parent) if f else ''
|
|
1042
|
+
if d and d != '.':
|
|
1043
|
+
dirs[d] = dirs.get(d, 0) + 1
|
|
1044
|
+
if dirs:
|
|
1045
|
+
lines += ["| Directory | Items |", "|-----------|-------|"]
|
|
1046
|
+
for d, count in sorted(dirs.items(), key=lambda x: -x[1]):
|
|
1047
|
+
lines.append(f"| `{d}` | {count} |")
|
|
1048
|
+
return '\n'.join(lines) + '\n'
|
|
1049
|
+
|
|
1050
|
+
|
|
1051
|
+
def build_background_jobs_section() -> str:
|
|
1052
|
+
lines = ["# Section 08 — Background Jobs\n\n"]
|
|
1053
|
+
jobs = []
|
|
1054
|
+
# Celery
|
|
1055
|
+
for f in PROJECT_ROOT.rglob('*.py'):
|
|
1056
|
+
if any(p in IGNORE_DIRS for p in f.parts):
|
|
1057
|
+
continue
|
|
1058
|
+
try:
|
|
1059
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
1060
|
+
for m in re.finditer(r'@(?:app|celery)\.task|@shared_task', source):
|
|
1061
|
+
# Find function after decorator
|
|
1062
|
+
fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
|
|
1063
|
+
if fn_m:
|
|
1064
|
+
line = source[:m.start()].count('\n') + 1
|
|
1065
|
+
jobs.append({'name': fn_m.group(1), 'type': 'celery', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
|
|
1066
|
+
except Exception:
|
|
1067
|
+
pass
|
|
1068
|
+
# Cron / APScheduler
|
|
1069
|
+
for f in PROJECT_ROOT.rglob('*.py'):
|
|
1070
|
+
if any(p in IGNORE_DIRS for p in f.parts):
|
|
1071
|
+
continue
|
|
1072
|
+
try:
|
|
1073
|
+
source = f.read_text(encoding='utf-8', errors='replace')
|
|
1074
|
+
for m in re.finditer(r'@scheduler\.scheduled_job|scheduler\.add_job|@cron', source):
|
|
1075
|
+
fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
|
|
1076
|
+
if fn_m:
|
|
1077
|
+
line = source[:m.start()].count('\n') + 1
|
|
1078
|
+
jobs.append({'name': fn_m.group(1), 'type': 'scheduler', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
|
|
1079
|
+
except Exception:
|
|
1080
|
+
pass
|
|
1081
|
+
|
|
1082
|
+
if not jobs:
|
|
1083
|
+
lines.append("_No background jobs detected._\n")
|
|
1084
|
+
else:
|
|
1085
|
+
lines += ["| Job | Type | File | Line |", "|-----|------|------|------|"]
|
|
1086
|
+
for j in jobs:
|
|
1087
|
+
lines.append(f"| `{j['name']}` | {j['type']} | {j['file']} | {j.get('line','')} |")
|
|
1088
|
+
return '\n'.join(lines) + '\n'
|
|
1089
|
+
|
|
1090
|
+
|
|
1091
|
+
def build_frontend_section(features: list[dict]) -> str:
|
|
1092
|
+
lines = ["# Section 09 — Frontend Features\n\n"]
|
|
1093
|
+
if not features:
|
|
1094
|
+
lines.append("_No frontend feature directories detected._\n")
|
|
1095
|
+
return '\n'.join(lines)
|
|
1096
|
+
lines += ["| Feature | Components | Hooks | Stores | API Files |",
|
|
1097
|
+
"|---------|-----------|-------|--------|-----------|"]
|
|
1098
|
+
for f in features:
|
|
1099
|
+
lines.append(
|
|
1100
|
+
f"| `{f['path']}` | {f['component_count']} | {f['hook_count']} "
|
|
1101
|
+
f"| {len(f.get('stores',[]))} | {len(f.get('api_files',[]))} |"
|
|
1102
|
+
)
|
|
1103
|
+
return '\n'.join(lines) + '\n'
|
|
1104
|
+
|
|
1105
|
+
|
|
1106
|
+
def build_tools_section(tools: list[dict]) -> str:
|
|
1107
|
+
lines = ["# Section 10 — Tools & Commands\n\n",
|
|
1108
|
+
"| Name | Command | Source |",
|
|
1109
|
+
"|------|---------|--------|"]
|
|
1110
|
+
for t in tools:
|
|
1111
|
+
lines.append(f"| `{t['name']}` | `{t['command']}` | {t.get('source','')} |")
|
|
1112
|
+
if not tools:
|
|
1113
|
+
lines.append("| _(no tools detected)_ | | |")
|
|
1114
|
+
return '\n'.join(lines) + '\n'
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
def build_migrations_section(migrations: list[dict]) -> str:
|
|
1118
|
+
lines = ["# Section 11 — Migrations\n\n"]
|
|
1119
|
+
if not migrations:
|
|
1120
|
+
lines.append("_No migrations detected._\n")
|
|
1121
|
+
return '\n'.join(lines)
|
|
1122
|
+
lines += ["| Revision | Tables Affected | Type | File |",
|
|
1123
|
+
"|----------|----------------|------|------|"]
|
|
1124
|
+
for m in migrations[-30:]: # last 30
|
|
1125
|
+
tables = ', '.join(m.get('tables_affected', []))[:60]
|
|
1126
|
+
lines.append(f"| `{m.get('revision','?')}` | {tables} | {m.get('type','?')} | {m.get('file','')} |")
|
|
1127
|
+
return '\n'.join(lines) + '\n'
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
def build_import_chains_section(chains: list[dict]) -> str:
|
|
1131
|
+
lines = ["# Section 12 — Import Chains\n\n",
|
|
1132
|
+
"_Traces route → service → model → table for key endpoints._\n\n"]
|
|
1133
|
+
if not chains:
|
|
1134
|
+
lines.append("_No import chains traced (requires Python source files with routes)._\n")
|
|
1135
|
+
return '\n'.join(lines)
|
|
1136
|
+
for c in chains:
|
|
1137
|
+
lines.append(f"**{c['route']}**")
|
|
1138
|
+
lines.append(f"```\n{c['chain']}\n```\n")
|
|
1139
|
+
return '\n'.join(lines)
|
|
1140
|
+
|
|
1141
|
+
|
|
1142
|
+
def build_frontend_backend_section(routes: list[dict], features: list[dict]) -> str:
|
|
1143
|
+
lines = ["# Section 13 — Frontend → Backend Map\n\n",
|
|
1144
|
+
"_Maps frontend API service calls to backend route paths._\n\n"]
|
|
1145
|
+
mappings = []
|
|
1146
|
+
api_calls: list[tuple[str,str]] = []
|
|
1147
|
+
|
|
1148
|
+
for feat in features:
|
|
1149
|
+
for api_file in feat.get('api_files', []):
|
|
1150
|
+
full = PROJECT_ROOT / feat['path'] / api_file
|
|
1151
|
+
if full.exists():
|
|
1152
|
+
try:
|
|
1153
|
+
source = full.read_text(encoding='utf-8', errors='replace')
|
|
1154
|
+
for m in re.finditer(r'[\'"`](/api/[^\'"` \n]+)', source):
|
|
1155
|
+
api_calls.append((m.group(1), str(full.relative_to(PROJECT_ROOT))))
|
|
1156
|
+
except Exception:
|
|
1157
|
+
pass
|
|
1158
|
+
|
|
1159
|
+
route_paths = {r.get('path',''): r for r in routes}
|
|
1160
|
+
for call, fe_file in api_calls:
|
|
1161
|
+
if call in route_paths:
|
|
1162
|
+
be = route_paths[call]
|
|
1163
|
+
mappings.append({'frontend': fe_file, 'url': call, 'backend': be.get('file','?')})
|
|
1164
|
+
|
|
1165
|
+
if not mappings:
|
|
1166
|
+
lines.append("_No frontend→backend mappings detected yet._\n")
|
|
1167
|
+
else:
|
|
1168
|
+
lines += ["| Frontend File | URL | Backend File |",
|
|
1169
|
+
"|--------------|-----|--------------|"]
|
|
1170
|
+
for m in mappings:
|
|
1171
|
+
lines.append(f"| {m['frontend']} | `{m['url']}` | {m['backend']} |")
|
|
1172
|
+
return '\n'.join(lines) + '\n'
|
|
1173
|
+
|
|
1174
|
+
|
|
1175
|
+
def build_proxy_section(proxy_rules: list[dict]) -> str:
|
|
1176
|
+
lines = ["# Section 14 — Reverse Proxy\n\n"]
|
|
1177
|
+
if not proxy_rules:
|
|
1178
|
+
lines.append("_No reverse proxy configuration detected._\n")
|
|
1179
|
+
return '\n'.join(lines)
|
|
1180
|
+
lines += ["| Type | Location | Upstream | File |",
|
|
1181
|
+
"|------|----------|----------|------|"]
|
|
1182
|
+
for r in proxy_rules:
|
|
1183
|
+
lines.append(f"| {r['type']} | `{r['location']}` | `{r['upstream']}` | {r['file']} |")
|
|
1184
|
+
return '\n'.join(lines) + '\n'
|
|
1185
|
+
|
|
1186
|
+
|
|
1187
|
+
def build_auth_section(auth_info: dict) -> str:
|
|
1188
|
+
lines = ["# Section 15 — Auth Configuration\n\n",
|
|
1189
|
+
f"**Provider**: {auth_info.get('provider','unknown')}\n\n"]
|
|
1190
|
+
files = auth_info.get('files', [])
|
|
1191
|
+
if files:
|
|
1192
|
+
lines.append("**Auth files:**")
|
|
1193
|
+
for f in files[:10]:
|
|
1194
|
+
lines.append(f"- `{f}`")
|
|
1195
|
+
patterns = auth_info.get('patterns', [])
|
|
1196
|
+
if patterns:
|
|
1197
|
+
lines.append(f"\n**Detected patterns**: {', '.join(f'`{p}`' for p in patterns[:10])}")
|
|
1198
|
+
return '\n'.join(lines) + '\n'
|
|
1199
|
+
|
|
1200
|
+
|
|
1201
|
+
def build_infra_section(stack: dict, services: list[dict]) -> str:
|
|
1202
|
+
lines = ["# Section 16 — Infrastructure Profile\n\n",
|
|
1203
|
+
f"| Property | Value |",
|
|
1204
|
+
"|----------|-------|",
|
|
1205
|
+
f"| **Language** | {stack.get('language','?')} |",
|
|
1206
|
+
f"| **Framework** | {stack.get('framework','?')} |",
|
|
1207
|
+
f"| **Database** | {stack.get('database','?')} |",
|
|
1208
|
+
f"| **ORM** | {stack.get('orm','?')} |",
|
|
1209
|
+
f"| **Auth** | {stack.get('auth','?')} |",
|
|
1210
|
+
f"| **Package Manager** | {stack.get('package_manager','?')} |",
|
|
1211
|
+
f"| **Infrastructure** | {stack.get('infrastructure','?')} |",
|
|
1212
|
+
f"| **Services** | {len(services)} docker services |",
|
|
1213
|
+
""]
|
|
1214
|
+
return '\n'.join(lines)
|
|
1215
|
+
|
|
1216
|
+
|
|
1217
|
+
def build_learned_vocab_section() -> str:
|
|
1218
|
+
lines = ["# Section 17 — Learned Vocabulary\n\n",
|
|
1219
|
+
"_Aliases mined from Claude Code session history. Score = frequency × recency._\n\n"]
|
|
1220
|
+
learned = {}
|
|
1221
|
+
if LEARNED_VOC.exists():
|
|
1222
|
+
try:
|
|
1223
|
+
learned = json.loads(LEARNED_VOC.read_text())
|
|
1224
|
+
except Exception:
|
|
1225
|
+
pass
|
|
1226
|
+
if not learned:
|
|
1227
|
+
lines.append("_No session-mined vocabulary yet. Accumulates over time._\n")
|
|
1228
|
+
return '\n'.join(lines)
|
|
1229
|
+
lines += ["| Alias | Score | Targets | Last Seen |",
|
|
1230
|
+
"|-------|-------|---------|-----------|"]
|
|
1231
|
+
for alias, data in sorted(learned.items(), key=lambda x: -x[1].get('score', 0)):
|
|
1232
|
+
if data.get('score', 0) >= 2:
|
|
1233
|
+
targets = ', '.join(data.get('targets', []))[:60]
|
|
1234
|
+
lines.append(f"| {alias} | {data.get('score',0):.1f} | {targets} | {data.get('last_seen','?')} |")
|
|
1235
|
+
return '\n'.join(lines) + '\n'
|
|
1236
|
+
|
|
1237
|
+
|
|
1238
|
+
def build_dead_code_section(candidates: list[dict]) -> str:
|
|
1239
|
+
lines = ["# Section 18 — Dead Code Candidates\n\n",
|
|
1240
|
+
"> Files flagged for human review: no recent commits AND low vocabulary score.\n",
|
|
1241
|
+
"> **Do not auto-delete.** Review before removing.\n\n"]
|
|
1242
|
+
if not candidates:
|
|
1243
|
+
lines.append("_No dead code candidates detected._\n")
|
|
1244
|
+
return '\n'.join(lines)
|
|
1245
|
+
for c in candidates:
|
|
1246
|
+
lines.append(f"- `{c['file']}` — {c.get('reason','')}")
|
|
1247
|
+
return '\n'.join(lines) + '\n'
|
|
1248
|
+
|
|
1249
|
+
|
|
1250
|
+
def build_doc_pointers_section() -> str:
|
|
1251
|
+
lines = ["# Section 19 — Documentation Pointers\n\n"]
|
|
1252
|
+
docs = []
|
|
1253
|
+
doc_dirs = ['docs', 'doc', 'documentation', 'wiki', '.docs']
|
|
1254
|
+
doc_exts = {'.md', '.rst', '.txt', '.adoc'}
|
|
1255
|
+
|
|
1256
|
+
for doc_dir in doc_dirs:
|
|
1257
|
+
d = PROJECT_ROOT / doc_dir
|
|
1258
|
+
if d.is_dir():
|
|
1259
|
+
for f in sorted(d.rglob('*')):
|
|
1260
|
+
if f.is_file() and f.suffix in doc_exts:
|
|
1261
|
+
docs.append(str(f.relative_to(PROJECT_ROOT)))
|
|
1262
|
+
|
|
1263
|
+
# Root-level docs
|
|
1264
|
+
for f in PROJECT_ROOT.glob('*.md'):
|
|
1265
|
+
docs.append(str(f.relative_to(PROJECT_ROOT)))
|
|
1266
|
+
|
|
1267
|
+
if not docs:
|
|
1268
|
+
lines.append("_No documentation files found._\n")
|
|
1269
|
+
else:
|
|
1270
|
+
for d in docs[:30]:
|
|
1271
|
+
lines.append(f"- [`{d}`]({d})")
|
|
1272
|
+
return '\n'.join(lines) + '\n'
|
|
1273
|
+
|
|
1274
|
+
|
|
1275
|
+
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
1276
|
+
# ║ PROJECT MAP TOC ║
|
|
1277
|
+
# ╚══════════════════════════════════════════════════════════════════════════╝
|
|
1278
|
+
|
|
1279
|
+
def build_project_map(
|
|
1280
|
+
stack: dict,
|
|
1281
|
+
routes: list[dict],
|
|
1282
|
+
models: list[dict],
|
|
1283
|
+
schemas: list[dict],
|
|
1284
|
+
features: list[dict],
|
|
1285
|
+
migrations: list[dict],
|
|
1286
|
+
services: list[dict],
|
|
1287
|
+
vocab: list[dict],
|
|
1288
|
+
section_files: list[tuple[str, Path]],
|
|
1289
|
+
) -> str:
|
|
1290
|
+
now = datetime.now().strftime('%Y-%m-%d %H:%M')
|
|
1291
|
+
name = stack.get('name', PROJECT_ROOT.name)
|
|
1292
|
+
slug = stack.get('slug', name)
|
|
1293
|
+
|
|
1294
|
+
lines = [
|
|
1295
|
+
f"# {name} — Project Map\n",
|
|
1296
|
+
f"> Auto-generated by `generate.py` on {now}. Do not edit manually.\n\n",
|
|
1297
|
+
f"## Stats\n",
|
|
1298
|
+
f"| Metric | Count |",
|
|
1299
|
+
f"|--------|-------|",
|
|
1300
|
+
f"| API Routes | {len(routes)} |",
|
|
1301
|
+
f"| Data Models | {len(models)} |",
|
|
1302
|
+
f"| Schemas/DTOs | {len(schemas)} |",
|
|
1303
|
+
f"| Frontend Features | {len(features)} |",
|
|
1304
|
+
f"| Migrations | {len(migrations)} |",
|
|
1305
|
+
f"| Docker Services | {len(services)} |",
|
|
1306
|
+
f"| Vocabulary Entries | {len(vocab)} |",
|
|
1307
|
+
f"| Stack | {stack.get('language','?')} / {stack.get('framework','?')} |",
|
|
1308
|
+
"",
|
|
1309
|
+
"## Section Index\n",
|
|
1310
|
+
"| # | Section | Size | When to Read |",
|
|
1311
|
+
"|---|---------|------|--------------|",
|
|
1312
|
+
]
|
|
1313
|
+
|
|
1314
|
+
WHEN_TO_READ = {
|
|
1315
|
+
'01': 'Any task — start here if you don\'t know where the code lives',
|
|
1316
|
+
'02': 'Debugging connectivity, adding a service, understanding ports',
|
|
1317
|
+
'03': 'Environment setup, missing vars, config issues',
|
|
1318
|
+
'04': 'Adding/editing API endpoints, checking what routes exist',
|
|
1319
|
+
'05': 'Changing database schema, adding fields, understanding relations',
|
|
1320
|
+
'06': 'Adding DTOs, changing request/response shapes',
|
|
1321
|
+
'07': 'Adding service logic, understanding service boundaries',
|
|
1322
|
+
'08': 'Working with background jobs, queues, scheduled tasks',
|
|
1323
|
+
'09': 'Frontend feature work, understanding UI structure',
|
|
1324
|
+
'10': 'Available commands, scripts, developer tooling',
|
|
1325
|
+
'11': 'Database migrations, schema history',
|
|
1326
|
+
'12': 'Tracing data flow from HTTP request to DB',
|
|
1327
|
+
'13': 'Understanding which frontend calls which backend endpoint',
|
|
1328
|
+
'14': 'Proxy routing, nginx/caddy config',
|
|
1329
|
+
'15': 'Auth flow, sessions, permissions',
|
|
1330
|
+
'16': 'Infrastructure overview, tech stack summary',
|
|
1331
|
+
'17': 'Vocabulary learned from past sessions',
|
|
1332
|
+
'18': 'Dead code review',
|
|
1333
|
+
'19': 'Finding documentation, READMEs, wikis',
|
|
1334
|
+
}
|
|
1335
|
+
|
|
1336
|
+
for name_part, path in section_files:
|
|
1337
|
+
num = name_part.split('-')[0]
|
|
1338
|
+
display = name_part.replace('-', ' ').title()
|
|
1339
|
+
size_kb = section_size_kb(path)
|
|
1340
|
+
when = WHEN_TO_READ.get(num, '')
|
|
1341
|
+
lines.append(f"| [{num}](sections/{path.name}) | {display} | {size_kb:.1f} KB | {when} |")
|
|
1342
|
+
|
|
1343
|
+
lines += [
|
|
1344
|
+
"",
|
|
1345
|
+
"## Quick Routing\n",
|
|
1346
|
+
"| Task | Read Sections |",
|
|
1347
|
+
"|------|--------------|",
|
|
1348
|
+
"| Feature / UX work | 01 → 09 → 04 |",
|
|
1349
|
+
"| Add model or field | 05 → 06 → 12 |",
|
|
1350
|
+
"| Troubleshoot error | 02 → 03 → 14 |",
|
|
1351
|
+
"| Infrastructure / scaling | 16 → 02 |",
|
|
1352
|
+
"| Auth / security | 15 → 19 |",
|
|
1353
|
+
"| What tools exist | 10 |",
|
|
1354
|
+
"| Background jobs | 08 |",
|
|
1355
|
+
"| Migration history | 11 |",
|
|
1356
|
+
"",
|
|
1357
|
+
f"## Regenerate\n",
|
|
1358
|
+
"```bash",
|
|
1359
|
+
"python .claude/project-map/generate.py # skip if unchanged",
|
|
1360
|
+
"python .claude/project-map/generate.py --force # always regenerate",
|
|
1361
|
+
"```",
|
|
1362
|
+
]
|
|
1363
|
+
return '\n'.join(lines) + '\n'
|
|
1364
|
+
|
|
1365
|
+
|
|
1366
|
+
# ╔══════════════════════════════════════════════════════════════════════════╗
|
|
1367
|
+
# ║ MAIN ║
|
|
1368
|
+
# ╚══════════════════════════════════════════════════════════════════════════╝
|
|
1369
|
+
|
|
1370
|
+
def main() -> None:
|
|
1371
|
+
parser = argparse.ArgumentParser(description='Babel Fish — generate project map')
|
|
1372
|
+
parser.add_argument('--force', action='store_true', help='Force regeneration even if checksums match')
|
|
1373
|
+
parser.add_argument('--project-root', type=Path, default=None, help='Override project root')
|
|
1374
|
+
parser.add_argument('--stack-json', type=Path, default=None, help='Path to stack.json from detect-stack.sh')
|
|
1375
|
+
args = parser.parse_args()
|
|
1376
|
+
|
|
1377
|
+
global PROJECT_ROOT
|
|
1378
|
+
if args.project_root:
|
|
1379
|
+
PROJECT_ROOT = args.project_root.resolve()
|
|
1380
|
+
|
|
1381
|
+
print(f"[generate] Project root: {PROJECT_ROOT}")
|
|
1382
|
+
|
|
1383
|
+
# 1. Checksum check
|
|
1384
|
+
watched = collect_watched_files()
|
|
1385
|
+
checksum = compute_checksum(watched)
|
|
1386
|
+
|
|
1387
|
+
if not args.force and is_unchanged(checksum):
|
|
1388
|
+
print("[generate] ✓ No changes detected — skipping regeneration (use --force to override)")
|
|
1389
|
+
sys.exit(0)
|
|
1390
|
+
|
|
1391
|
+
# 2. Load stack
|
|
1392
|
+
stack = load_stack(args.stack_json)
|
|
1393
|
+
print(f"[generate] Stack: {stack['language']} / {stack['framework']}")
|
|
1394
|
+
|
|
1395
|
+
# 3. Collect all relevant files
|
|
1396
|
+
py_files = [f for f in watched if f.suffix == '.py']
|
|
1397
|
+
ts_files = [f for f in watched if f.suffix in ('.ts','.tsx','.js','.jsx')]
|
|
1398
|
+
go_files = [f for f in watched if f.suffix == '.go']
|
|
1399
|
+
|
|
1400
|
+
# 4. Parse
|
|
1401
|
+
print("[generate] Parsing routes...")
|
|
1402
|
+
routes: list[dict] = []
|
|
1403
|
+
if stack['language'] in ('python', 'unknown'):
|
|
1404
|
+
routes.extend(PythonRouteParser().parse(py_files))
|
|
1405
|
+
if stack['language'] in ('typescript', 'javascript', 'unknown'):
|
|
1406
|
+
routes.extend(TSRouteParser().parse(ts_files))
|
|
1407
|
+
if stack['language'] in ('go', 'unknown'):
|
|
1408
|
+
routes.extend(GoRouteParser().parse(go_files))
|
|
1409
|
+
|
|
1410
|
+
print("[generate] Parsing models...")
|
|
1411
|
+
models: list[dict] = []
|
|
1412
|
+
if stack['language'] in ('python', 'unknown'):
|
|
1413
|
+
models.extend(PythonModelParser().parse(py_files))
|
|
1414
|
+
if stack['language'] in ('typescript', 'javascript', 'unknown'):
|
|
1415
|
+
models.extend(TSModelParser().parse(ts_files + [f for f in watched if f.suffix == '.prisma']))
|
|
1416
|
+
if stack['language'] in ('go', 'unknown'):
|
|
1417
|
+
models.extend(GoModelParser().parse(go_files))
|
|
1418
|
+
|
|
1419
|
+
print("[generate] Parsing schemas...")
|
|
1420
|
+
schemas = PythonSchemaParser().parse(py_files)
|
|
1421
|
+
|
|
1422
|
+
print("[generate] Scanning environment...")
|
|
1423
|
+
services = DockerComposeParser().parse()
|
|
1424
|
+
env_entries = EnvParser().parse()
|
|
1425
|
+
migrations = MigrationParser().parse()
|
|
1426
|
+
features = FrontendScanner().scan()
|
|
1427
|
+
tools = ToolsScanner().scan()
|
|
1428
|
+
auth_info = AuthScanner().scan()
|
|
1429
|
+
proxy_rules = ReverseProxyScanner().scan()
|
|
1430
|
+
|
|
1431
|
+
print("[generate] Building vocabulary...")
|
|
1432
|
+
vocab = VocabularyBuilder().build(routes, models, schemas, features, stack)
|
|
1433
|
+
|
|
1434
|
+
print("[generate] Tracing import chains...")
|
|
1435
|
+
chains = ImportChainTracer().trace(routes)
|
|
1436
|
+
|
|
1437
|
+
print("[generate] Detecting dead code candidates...")
|
|
1438
|
+
dead_code = DeadCodeDetector().detect(vocab)
|
|
1439
|
+
|
|
1440
|
+
# 5. Write sections
|
|
1441
|
+
print("[generate] Writing sections...")
|
|
1442
|
+
section_files: list[tuple[str, Path]] = [
|
|
1443
|
+
('01-vocabulary', write_section('01', 'vocabulary', build_vocabulary_section(vocab))),
|
|
1444
|
+
('02-service-topology', write_section('02', 'service-topology', build_topology_section(services))),
|
|
1445
|
+
('03-environment', write_section('03', 'environment', build_environment_section(env_entries))),
|
|
1446
|
+
('04-api-routes', write_section('04', 'api-routes', build_routes_section(routes))),
|
|
1447
|
+
('05-data-models', write_section('05', 'data-models', build_models_section(models))),
|
|
1448
|
+
('06-schemas', write_section('06', 'schemas', build_schemas_section(schemas))),
|
|
1449
|
+
('07-services', write_section('07', 'services', build_services_section(routes, models))),
|
|
1450
|
+
('08-background-jobs', write_section('08', 'background-jobs', build_background_jobs_section())),
|
|
1451
|
+
('09-frontend-features', write_section('09', 'frontend-features', build_frontend_section(features))),
|
|
1452
|
+
('10-tools-commands', write_section('10', 'tools-commands', build_tools_section(tools))),
|
|
1453
|
+
('11-migrations', write_section('11', 'migrations', build_migrations_section(migrations))),
|
|
1454
|
+
('12-import-chains', write_section('12', 'import-chains', build_import_chains_section(chains))),
|
|
1455
|
+
('13-frontend-backend-map',write_section('13', 'frontend-backend-map',build_frontend_backend_section(routes, features))),
|
|
1456
|
+
('14-reverse-proxy', write_section('14', 'reverse-proxy', build_proxy_section(proxy_rules))),
|
|
1457
|
+
('15-auth-config', write_section('15', 'auth-config', build_auth_section(auth_info))),
|
|
1458
|
+
('16-infra-profile', write_section('16', 'infra-profile', build_infra_section(stack, services))),
|
|
1459
|
+
('17-learned-vocabulary', write_section('17', 'learned-vocabulary', build_learned_vocab_section())),
|
|
1460
|
+
('18-dead-code', write_section('18', 'dead-code', build_dead_code_section(dead_code))),
|
|
1461
|
+
('19-doc-pointers', write_section('19', 'doc-pointers', build_doc_pointers_section())),
|
|
1462
|
+
]
|
|
1463
|
+
|
|
1464
|
+
# 6. Write PROJECT_MAP.md
|
|
1465
|
+
print("[generate] Writing PROJECT_MAP.md...")
|
|
1466
|
+
project_map = build_project_map(stack, routes, models, schemas, features, migrations, services, vocab, section_files)
|
|
1467
|
+
(MAP_DIR / 'PROJECT_MAP.md').write_text(project_map, encoding='utf-8')
|
|
1468
|
+
|
|
1469
|
+
# 7. Update checksums
|
|
1470
|
+
save_checksums({'input_hash': checksum, 'generated_at': datetime.now().isoformat(), 'route_count': len(routes), 'model_count': len(models)})
|
|
1471
|
+
|
|
1472
|
+
# 8. Ensure learned-vocabulary.json exists
|
|
1473
|
+
if not LEARNED_VOC.exists():
|
|
1474
|
+
LEARNED_VOC.write_text('{}', encoding='utf-8')
|
|
1475
|
+
|
|
1476
|
+
total_kb = sum(section_size_kb(p) for _, p in section_files)
|
|
1477
|
+
print(f"[generate] ✓ Done — {len(routes)} routes, {len(models)} models, {len(vocab)} vocab entries, {total_kb:.1f} KB total")
|
|
1478
|
+
|
|
1479
|
+
|
|
1480
|
+
if __name__ == '__main__':
|
|
1481
|
+
main()
|