@theglitchking/babel-fish 1.0.2 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/.claude/install.sh +666 -0
  2. package/.claude/project-map/PROJECT_MAP.md +61 -0
  3. package/.claude/project-map/checksums.json +6 -0
  4. package/.claude/project-map/generate.py +1481 -0
  5. package/.claude/project-map/grader.py +583 -0
  6. package/.claude/project-map/learned-vocabulary.json +1 -0
  7. package/.claude/project-map/mine-sessions.py +419 -0
  8. package/.claude/project-map/reports/install-report.md +54 -0
  9. package/.claude/project-map/reports/iteration-01-report.md +54 -0
  10. package/.claude/project-map/reports/iteration-01-score.json +7 -0
  11. package/.claude/project-map/sections/01-vocabulary.md +8 -0
  12. package/.claude/project-map/sections/02-service-topology.md +6 -0
  13. package/.claude/project-map/sections/03-environment.md +6 -0
  14. package/.claude/project-map/sections/04-api-routes.md +6 -0
  15. package/.claude/project-map/sections/05-data-models.md +4 -0
  16. package/.claude/project-map/sections/06-schemas.md +4 -0
  17. package/.claude/project-map/sections/07-services.md +6 -0
  18. package/.claude/project-map/sections/08-background-jobs.md +5 -0
  19. package/.claude/project-map/sections/09-frontend-features.md +4 -0
  20. package/.claude/project-map/sections/10-tools-commands.md +8 -0
  21. package/.claude/project-map/sections/11-migrations.md +4 -0
  22. package/.claude/project-map/sections/12-import-chains.md +7 -0
  23. package/.claude/project-map/sections/13-frontend-backend-map.md +8 -0
  24. package/.claude/project-map/sections/14-reverse-proxy.md +4 -0
  25. package/.claude/project-map/sections/15-auth-config.md +6 -0
  26. package/.claude/project-map/sections/16-infra-profile.md +13 -0
  27. package/.claude/project-map/sections/17-learned-vocabulary.md +7 -0
  28. package/.claude/project-map/sections/18-dead-code.md +9 -0
  29. package/.claude/project-map/sections/19-doc-pointers.md +5 -0
  30. package/.claude/project-map/stack.json +12 -0
  31. package/.claude/rules/operational-runbook.md +40 -0
  32. package/.claude/rules/project-vocabulary.md +25 -0
  33. package/.claude/scripts/detect-stack.sh +222 -0
  34. package/.claude/scripts/ensure-python.sh +100 -0
  35. package/.claude/scripts/statusline.sh +27 -0
  36. package/.claude/scripts/validate.sh +59 -0
  37. package/.claude/settings.json +6 -0
  38. package/.claude/skills/babel-fish-developer-skill/SKILL.md +56 -0
  39. package/.claude/templates/SKILL.md.template +56 -0
  40. package/.claude/templates/operational-runbook.md.template +40 -0
  41. package/.claude/templates/project-vocabulary.md.template +25 -0
  42. package/.claude-plugin/marketplace.json +2 -2
  43. package/.claude-plugin/plugin.json +1 -1
  44. package/.githooks/install.sh +4 -0
  45. package/.githooks/pre-commit +22 -0
  46. package/CHANGELOG.md +72 -0
  47. package/README.md +18 -0
  48. package/bin/babel-fish.js +78 -71
  49. package/commands/policy.md +16 -0
  50. package/commands/relink.md +6 -0
  51. package/commands/status.md +6 -0
  52. package/commands/update.md +6 -0
  53. package/hooks/hooks.json +15 -0
  54. package/hooks/session-start.js +11 -0
  55. package/package.json +19 -3
  56. package/scripts/link-skills.js +31 -0
@@ -0,0 +1,1481 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ generate.py — Babel Fish
4
+ Stack-agnostic introspection script for the babel-fish codebase mapper plugin.
5
+ Produces a split-section project map under .claude/project-map/sections/
6
+
7
+ Usage:
8
+ python generate.py [--force] [--project-root PATH] [--stack-json PATH]
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import ast
13
+ import argparse
14
+ import hashlib
15
+ import json
16
+ import os
17
+ import re
18
+ import subprocess
19
+ import sys
20
+ from datetime import datetime
21
+ from pathlib import Path
22
+ from typing import Any
23
+
24
+ # ── Try optional deps ────────────────────────────────────────────────────────
25
+ try:
26
+ import yaml
27
+ HAS_YAML = True
28
+ except ImportError:
29
+ HAS_YAML = False
30
+
31
+ # ── Paths ────────────────────────────────────────────────────────────────────
32
+ SCRIPT_DIR = Path(__file__).parent
33
+ MAP_DIR = SCRIPT_DIR
34
+ SECTIONS_DIR = MAP_DIR / "sections"
35
+ CHECKSUMS = MAP_DIR / "checksums.json"
36
+ LEARNED_VOC = MAP_DIR / "learned-vocabulary.json"
37
+
38
+ # Resolve project root: two levels up from .claude/project-map/
39
+ PROJECT_ROOT = MAP_DIR.parent.parent
40
+
41
+ SECTIONS_DIR.mkdir(parents=True, exist_ok=True)
42
+
43
+ # ── Secrets guard ────────────────────────────────────────────────────────────
44
+ SECRET_PATTERNS = re.compile(
45
+ r'(?i)(password|secret|token|api_key|apikey|private_key|auth_token|'
46
+ r'access_key|secret_key|client_secret|db_pass|database_password|'
47
+ r'stripe_key|twilio_auth|sendgrid_key|aws_secret)\s*[=:]\s*\S+'
48
+ )
49
+
50
+ def redact_secrets(text: str) -> str:
51
+ return SECRET_PATTERNS.sub(r'[REDACTED]', text)
52
+
53
+
54
+ # ── Checksum logic ───────────────────────────────────────────────────────────
55
+ WATCHED_EXTENSIONS = {
56
+ '.py', '.ts', '.tsx', '.js', '.jsx', '.go', '.java', '.kt',
57
+ '.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env',
58
+ }
59
+ WATCHED_NAMES = {
60
+ 'docker-compose.yml', 'docker-compose.yaml', 'docker-compose.dev.yml',
61
+ 'package.json', 'requirements.txt', 'pyproject.toml', 'go.mod',
62
+ 'Cargo.toml', 'pom.xml', 'Gemfile', 'Makefile',
63
+ }
64
+ IGNORE_DIRS = {
65
+ '.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
66
+ 'dist', 'build', '.next', '.nuxt', 'target', 'vendor', '.cache',
67
+ '.claude', '.planning',
68
+ }
69
+
70
+ def collect_watched_files() -> list[Path]:
71
+ result = []
72
+ for path in sorted(PROJECT_ROOT.rglob('*')):
73
+ if any(p in IGNORE_DIRS for p in path.parts):
74
+ continue
75
+ if path.is_file() and (path.suffix in WATCHED_EXTENSIONS or path.name in WATCHED_NAMES):
76
+ result.append(path)
77
+ return result
78
+
79
+ def compute_checksum(files: list[Path]) -> str:
80
+ h = hashlib.sha256()
81
+ for f in files:
82
+ h.update(str(f).encode())
83
+ try:
84
+ h.update(str(f.stat().st_mtime_ns).encode())
85
+ except OSError:
86
+ pass
87
+ return h.hexdigest()
88
+
89
+ def load_checksums() -> dict:
90
+ if CHECKSUMS.exists():
91
+ try:
92
+ return json.loads(CHECKSUMS.read_text())
93
+ except Exception:
94
+ pass
95
+ return {}
96
+
97
+ def save_checksums(data: dict) -> None:
98
+ CHECKSUMS.write_text(json.dumps(data, indent=2))
99
+
100
+ def is_unchanged(checksum: str) -> bool:
101
+ stored = load_checksums()
102
+ return stored.get('input_hash') == checksum
103
+
104
+
105
+ # ── Stack detection (reads stack.json if present, else fallback) ─────────────
106
+ def load_stack(stack_json_path: Path | None = None) -> dict:
107
+ candidates = [
108
+ stack_json_path,
109
+ MAP_DIR / "stack.json",
110
+ PROJECT_ROOT / ".claude" / "stack.json",
111
+ ]
112
+ for path in candidates:
113
+ if path and path.exists():
114
+ try:
115
+ return json.loads(path.read_text())
116
+ except Exception:
117
+ pass
118
+ return {
119
+ "name": PROJECT_ROOT.name,
120
+ "slug": re.sub(r'[^a-z0-9]', '-', PROJECT_ROOT.name.lower()).strip('-'),
121
+ "language": "unknown",
122
+ "framework": "unknown",
123
+ "database": "unknown",
124
+ "orm": "unknown",
125
+ "auth": "unknown",
126
+ "package_manager": "unknown",
127
+ "infrastructure": "none",
128
+ "project_root": str(PROJECT_ROOT),
129
+ }
130
+
131
+
132
+ # ╔══════════════════════════════════════════════════════════════════════════╗
133
+ # ║ PARSERS ║
134
+ # ╚══════════════════════════════════════════════════════════════════════════╝
135
+
136
+ # ── Python / FastAPI / Django / Flask ────────────────────────────────────────
137
+
138
+ class PythonRouteParser:
139
+ """Extract routes from Python routers using ast."""
140
+
141
+ DECORATOR_PATTERNS = re.compile(
142
+ r'@(router|app|api_router|blueprint)\.(get|post|put|patch|delete|head|options|websocket)\s*\('
143
+ )
144
+
145
+ def parse(self, files: list[Path]) -> list[dict]:
146
+ routes = []
147
+ for f in files:
148
+ if f.suffix != '.py':
149
+ continue
150
+ try:
151
+ source = f.read_text(encoding='utf-8', errors='replace')
152
+ tree = ast.parse(source, filename=str(f))
153
+ routes.extend(self._extract_routes(tree, f, source))
154
+ except SyntaxError:
155
+ pass
156
+ return routes
157
+
158
+ def _extract_routes(self, tree: ast.AST, filepath: Path, source: str) -> list[dict]:
159
+ routes = []
160
+ lines = source.splitlines()
161
+ for node in ast.walk(tree):
162
+ if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
163
+ continue
164
+ for decorator in node.decorator_list:
165
+ info = self._parse_decorator(decorator, lines, node.lineno)
166
+ if info:
167
+ info['file'] = str(filepath.relative_to(PROJECT_ROOT))
168
+ info['line'] = node.lineno
169
+ info['function'] = node.name
170
+ routes.append(info)
171
+ return routes
172
+
173
+ def _parse_decorator(self, decorator: ast.expr, lines: list[str], lineno: int) -> dict | None:
174
+ if isinstance(decorator, ast.Call):
175
+ func = decorator.func
176
+ if isinstance(func, ast.Attribute) and func.attr in (
177
+ 'get','post','put','patch','delete','head','options','websocket'
178
+ ):
179
+ method = func.attr.upper()
180
+ path = ''
181
+ if decorator.args:
182
+ arg = decorator.args[0]
183
+ if isinstance(arg, ast.Constant):
184
+ path = str(arg.value)
185
+ return {'method': method, 'path': path}
186
+ return None
187
+
188
+
189
+ class PythonModelParser:
190
+ """Extract SQLAlchemy / Django ORM models using ast."""
191
+
192
+ def parse(self, files: list[Path]) -> list[dict]:
193
+ models = []
194
+ for f in files:
195
+ if f.suffix != '.py':
196
+ continue
197
+ try:
198
+ source = f.read_text(encoding='utf-8', errors='replace')
199
+ tree = ast.parse(source, filename=str(f))
200
+ models.extend(self._extract_models(tree, f))
201
+ except SyntaxError:
202
+ pass
203
+ return models
204
+
205
+ def _extract_models(self, tree: ast.AST, filepath: Path) -> list[dict]:
206
+ models = []
207
+ for node in ast.walk(tree):
208
+ if not isinstance(node, ast.ClassDef):
209
+ continue
210
+ bases = [self._base_name(b) for b in node.bases]
211
+ is_model = any(
212
+ b in ('Base', 'Model', 'BaseModel', 'DeclarativeBase', 'AbstractModel')
213
+ or 'Model' in b
214
+ for b in bases if b
215
+ )
216
+ if not is_model:
217
+ continue
218
+ columns = self._extract_columns(node)
219
+ models.append({
220
+ 'name': node.name,
221
+ 'file': str(filepath.relative_to(PROJECT_ROOT)),
222
+ 'line': node.lineno,
223
+ 'bases': bases,
224
+ 'columns': columns,
225
+ })
226
+ return models
227
+
228
+ def _base_name(self, node: ast.expr) -> str:
229
+ if isinstance(node, ast.Name):
230
+ return node.id
231
+ if isinstance(node, ast.Attribute):
232
+ return node.attr
233
+ return ''
234
+
235
+ def _extract_columns(self, class_node: ast.ClassDef) -> list[dict]:
236
+ cols = []
237
+ for node in ast.walk(class_node):
238
+ if isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name):
239
+ cols.append({
240
+ 'name': node.target.id,
241
+ 'type': ast.unparse(node.annotation) if hasattr(ast, 'unparse') else '?',
242
+ })
243
+ elif isinstance(node, ast.Assign):
244
+ for target in node.targets:
245
+ if isinstance(target, ast.Name):
246
+ if isinstance(node.value, ast.Call):
247
+ func_name = ''
248
+ if isinstance(node.value.func, ast.Name):
249
+ func_name = node.value.func.id
250
+ elif isinstance(node.value.func, ast.Attribute):
251
+ func_name = node.value.func.attr
252
+ if func_name in ('Column', 'Field', 'CharField', 'IntegerField',
253
+ 'TextField', 'BooleanField', 'ForeignKey',
254
+ 'ManyToManyField', 'DateTimeField', 'mapped_column'):
255
+ cols.append({'name': target.id, 'type': func_name})
256
+ return cols
257
+
258
+
259
+ class PythonSchemaParser:
260
+ """Extract Pydantic schemas / Django serializers using ast."""
261
+
262
+ def parse(self, files: list[Path]) -> list[dict]:
263
+ schemas = []
264
+ for f in files:
265
+ if f.suffix != '.py':
266
+ continue
267
+ try:
268
+ source = f.read_text(encoding='utf-8', errors='replace')
269
+ tree = ast.parse(source, filename=str(f))
270
+ schemas.extend(self._extract_schemas(tree, f))
271
+ except SyntaxError:
272
+ pass
273
+ return schemas
274
+
275
+ def _extract_schemas(self, tree: ast.AST, filepath: Path) -> list[dict]:
276
+ schemas = []
277
+ for node in ast.walk(tree):
278
+ if not isinstance(node, ast.ClassDef):
279
+ continue
280
+ bases = [self._base_name(b) for b in node.bases]
281
+ is_schema = any(
282
+ b in ('BaseModel', 'Schema', 'Serializer', 'ModelSerializer',
283
+ 'TypedDict', 'NamedTuple')
284
+ or 'Schema' in b or 'Serializer' in b
285
+ for b in bases if b
286
+ )
287
+ if not is_schema:
288
+ continue
289
+ fields = [
290
+ n.target.id
291
+ for n in ast.walk(node)
292
+ if isinstance(n, ast.AnnAssign) and isinstance(n.target, ast.Name)
293
+ ]
294
+ schemas.append({
295
+ 'name': node.name,
296
+ 'file': str(filepath.relative_to(PROJECT_ROOT)),
297
+ 'line': node.lineno,
298
+ 'fields': fields,
299
+ })
300
+ return schemas
301
+
302
+ def _base_name(self, node: ast.expr) -> str:
303
+ if isinstance(node, ast.Name):
304
+ return node.id
305
+ if isinstance(node, ast.Attribute):
306
+ return node.attr
307
+ return ''
308
+
309
+
310
+ # ── TypeScript / JavaScript ──────────────────────────────────────────────────
311
+
312
+ class TSRouteParser:
313
+ """Extract routes from Next.js, Express, NestJS via regex."""
314
+
315
+ NEXT_APP_ROUTE = re.compile(r'export\s+async\s+function\s+(GET|POST|PUT|PATCH|DELETE|HEAD)\s*\(')
316
+ NEXT_PAGES_ROUTE = re.compile(r'export\s+default\s+(?:async\s+)?function\s+handler')
317
+ EXPRESS_ROUTE = re.compile(r'(?:router|app)\.(get|post|put|patch|delete)\s*\(\s*[\'"]([^\'"]+)[\'"]')
318
+ NEST_DECORATOR = re.compile(r'@(Get|Post|Put|Patch|Delete)\s*\(\s*(?:[\'"]([^\'"]*)[\'"])?\s*\)')
319
+
320
+ def parse(self, files: list[Path]) -> list[dict]:
321
+ routes = []
322
+ for f in files:
323
+ if f.suffix not in ('.ts', '.tsx', '.js', '.jsx'):
324
+ continue
325
+ try:
326
+ source = f.read_text(encoding='utf-8', errors='replace')
327
+ except Exception:
328
+ continue
329
+ rel = str(f.relative_to(PROJECT_ROOT))
330
+
331
+ # Next.js App Router
332
+ for m in self.NEXT_APP_ROUTE.finditer(source):
333
+ line = source[:m.start()].count('\n') + 1
334
+ path = self._infer_next_path(f)
335
+ routes.append({'method': m.group(1), 'path': path, 'file': rel, 'line': line})
336
+
337
+ # Express
338
+ for m in self.EXPRESS_ROUTE.finditer(source):
339
+ line = source[:m.start()].count('\n') + 1
340
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
341
+
342
+ # NestJS
343
+ for m in self.NEST_DECORATOR.finditer(source):
344
+ line = source[:m.start()].count('\n') + 1
345
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2) or '/', 'file': rel, 'line': line})
346
+
347
+ return routes
348
+
349
+ def _infer_next_path(self, f: Path) -> str:
350
+ """Infer URL path from Next.js file location."""
351
+ parts = f.parts
352
+ try:
353
+ # app router: after 'app/'
354
+ app_idx = parts.index('app') if 'app' in parts else None
355
+ if app_idx:
356
+ seg = parts[app_idx+1:]
357
+ seg = [s for s in seg if not s.startswith('(') and s not in ('route.ts','route.js','page.tsx','page.ts')]
358
+ return '/' + '/'.join(seg) if seg else '/'
359
+ # pages router
360
+ pages_idx = parts.index('pages') if 'pages' in parts else None
361
+ if pages_idx:
362
+ seg = list(parts[pages_idx+1:])
363
+ seg[-1] = re.sub(r'\.(ts|tsx|js|jsx)$', '', seg[-1])
364
+ if seg[-1] in ('index',):
365
+ seg = seg[:-1]
366
+ return '/' + '/'.join(seg) if seg else '/'
367
+ except (ValueError, IndexError):
368
+ pass
369
+ return f'/{f.stem}'
370
+
371
+
372
+ class TSModelParser:
373
+ """Extract Prisma schema entities and TypeORM entities via regex."""
374
+
375
+ PRISMA_MODEL = re.compile(r'^model\s+(\w+)\s*\{([^}]+)\}', re.MULTILINE)
376
+ PRISMA_FIELD = re.compile(r'^\s+(\w+)\s+(\w+)', re.MULTILINE)
377
+ TYPEORM_ENTITY = re.compile(r'@Entity\s*\(')
378
+ CLASS_NAME = re.compile(r'class\s+(\w+)')
379
+
380
+ def parse(self, files: list[Path]) -> list[dict]:
381
+ models = []
382
+ for f in files:
383
+ rel = str(f.relative_to(PROJECT_ROOT))
384
+ # Prisma
385
+ if f.suffix == '.prisma':
386
+ try:
387
+ source = f.read_text(encoding='utf-8', errors='replace')
388
+ for m in self.PRISMA_MODEL.finditer(source):
389
+ fields = [
390
+ {'name': fm.group(1), 'type': fm.group(2)}
391
+ for fm in self.PRISMA_FIELD.finditer(m.group(2))
392
+ ]
393
+ models.append({'name': m.group(1), 'file': rel, 'columns': fields})
394
+ except Exception:
395
+ pass
396
+ # TypeORM / NestJS entities
397
+ elif f.suffix in ('.ts', '.tsx'):
398
+ try:
399
+ source = f.read_text(encoding='utf-8', errors='replace')
400
+ if self.TYPEORM_ENTITY.search(source):
401
+ cm = self.CLASS_NAME.search(source)
402
+ if cm:
403
+ models.append({'name': cm.group(1), 'file': rel, 'columns': []})
404
+ except Exception:
405
+ pass
406
+ return models
407
+
408
+
409
+ # ── Go ───────────────────────────────────────────────────────────────────────
410
+
411
+ class GoRouteParser:
412
+ GIN = re.compile(r'(?:router|r|v\d+|api)\.(GET|POST|PUT|PATCH|DELETE)\s*\(\s*"([^"]+)"')
413
+ CHI = re.compile(r'r\.(Get|Post|Put|Patch|Delete)\s*\(\s*"([^"]+)"')
414
+ STD = re.compile(r'(?:mux|http)\.HandleFunc\s*\(\s*"([^"]+)"')
415
+
416
+ def parse(self, files: list[Path]) -> list[dict]:
417
+ routes = []
418
+ for f in files:
419
+ if f.suffix != '.go':
420
+ continue
421
+ try:
422
+ source = f.read_text(encoding='utf-8', errors='replace')
423
+ except Exception:
424
+ continue
425
+ rel = str(f.relative_to(PROJECT_ROOT))
426
+ for m in self.GIN.finditer(source):
427
+ line = source[:m.start()].count('\n') + 1
428
+ routes.append({'method': m.group(1), 'path': m.group(2), 'file': rel, 'line': line})
429
+ for m in self.CHI.finditer(source):
430
+ line = source[:m.start()].count('\n') + 1
431
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
432
+ for m in self.STD.finditer(source):
433
+ line = source[:m.start()].count('\n') + 1
434
+ routes.append({'method': 'ANY', 'path': m.group(1), 'file': rel, 'line': line})
435
+ return routes
436
+
437
+
438
+ class GoModelParser:
439
+ STRUCT = re.compile(r'type\s+(\w+)\s+struct\s*\{([^}]+)\}', re.DOTALL)
440
+ FIELD = re.compile(r'^\s+(\w+)\s+(\S+)', re.MULTILINE)
441
+
442
+ def parse(self, files: list[Path]) -> list[dict]:
443
+ models = []
444
+ for f in files:
445
+ if f.suffix != '.go':
446
+ continue
447
+ try:
448
+ source = f.read_text(encoding='utf-8', errors='replace')
449
+ except Exception:
450
+ continue
451
+ rel = str(f.relative_to(PROJECT_ROOT))
452
+ for m in self.STRUCT.finditer(source):
453
+ # Only include structs that look like data models (have db/json tags)
454
+ body = m.group(2)
455
+ if 'db:' in body or 'json:' in body or 'gorm:' in body:
456
+ fields = [
457
+ {'name': fm.group(1), 'type': fm.group(2)}
458
+ for fm in self.FIELD.finditer(body)
459
+ if not fm.group(1).startswith('//')
460
+ ]
461
+ models.append({'name': m.group(1), 'file': rel, 'columns': fields})
462
+ return models
463
+
464
+
465
+ # ── Docker Compose ───────────────────────────────────────────────────────────
466
+
467
+ class DockerComposeParser:
468
+ def parse(self) -> list[dict]:
469
+ services = []
470
+ candidates = [
471
+ 'docker-compose.yml', 'docker-compose.yaml',
472
+ 'docker-compose.dev.yml', 'docker-compose.development.yml',
473
+ 'docker-compose.prod.yml', 'docker-compose.production.yml',
474
+ ]
475
+ for name in candidates:
476
+ path = PROJECT_ROOT / name
477
+ if not path.exists():
478
+ continue
479
+ try:
480
+ data = self._load(path)
481
+ if data and isinstance(data.get('services'), dict):
482
+ for svc_name, svc in data['services'].items():
483
+ services.append({
484
+ 'name': svc_name,
485
+ 'image': svc.get('image', svc.get('build', '(build)')),
486
+ 'ports': svc.get('ports', []),
487
+ 'depends_on': svc.get('depends_on', []),
488
+ 'environment_keys': list(svc.get('environment', {}).keys())
489
+ if isinstance(svc.get('environment'), dict)
490
+ else [],
491
+ 'source_file': name,
492
+ })
493
+ except Exception:
494
+ pass
495
+ return services
496
+
497
+ def _load(self, path: Path) -> dict | None:
498
+ text = path.read_text(encoding='utf-8', errors='replace')
499
+ if HAS_YAML:
500
+ return yaml.safe_load(text)
501
+ # Regex fallback: extract service names at minimum
502
+ services: dict = {'services': {}}
503
+ in_services = False
504
+ indent = 0
505
+ for line in text.splitlines():
506
+ if re.match(r'^services\s*:', line):
507
+ in_services = True
508
+ continue
509
+ if in_services:
510
+ m = re.match(r'^(\s+)(\w[\w-]*)\s*:', line)
511
+ if m:
512
+ lvl = len(m.group(1))
513
+ if indent == 0:
514
+ indent = lvl
515
+ if lvl == indent:
516
+ services['services'][m.group(2)] = {}
517
+ return services
518
+
519
+
520
+ # ── Env Parser ───────────────────────────────────────────────────────────────
521
+
522
+ class EnvParser:
523
+ SECRET_KEYS = re.compile(
524
+ r'(?i)(password|secret|token|key|auth|credential|private|cert)'
525
+ )
526
+
527
+ def parse(self) -> list[dict]:
528
+ entries = []
529
+ candidates = ['.env', '.env.example', '.env.local', '.env.development', '.env.sample']
530
+ for name in candidates:
531
+ path = PROJECT_ROOT / name
532
+ if not path.exists():
533
+ continue
534
+ for line in path.read_text(encoding='utf-8', errors='replace').splitlines():
535
+ line = line.strip()
536
+ if not line or line.startswith('#'):
537
+ continue
538
+ if '=' not in line:
539
+ continue
540
+ key, _, raw_val = line.partition('=')
541
+ key = key.strip()
542
+ val = raw_val.strip()
543
+ is_secret = bool(self.SECRET_KEYS.search(key))
544
+ entries.append({
545
+ 'key': key,
546
+ 'value': '[SECRET]' if is_secret else val[:80],
547
+ 'is_secret': is_secret,
548
+ 'source': name,
549
+ })
550
+ return entries
551
+
552
+
553
+ # ── Migration Parser ─────────────────────────────────────────────────────────
554
+
555
+ class MigrationParser:
556
+ def parse(self) -> list[dict]:
557
+ migrations = []
558
+ # Alembic
559
+ alembic_dirs = list(PROJECT_ROOT.rglob('versions'))
560
+ for d in alembic_dirs:
561
+ if not d.is_dir():
562
+ continue
563
+ for f in sorted(d.glob('*.py')):
564
+ try:
565
+ source = f.read_text(encoding='utf-8', errors='replace')
566
+ rev = re.search(r'revision\s*=\s*[\'"]([^\'"]+)[\'"]', source)
567
+ down = re.search(r'down_revision\s*=\s*[\'"]?([^\'")\n]+)[\'"]?', source)
568
+ msg = re.search(r'Create Date.*\n.*"""(.+?)"""', source, re.DOTALL)
569
+ tables = re.findall(r'op\.(?:create_table|drop_table|add_column)\s*\(\s*[\'"]([^\'"]+)[\'"]', source)
570
+ migrations.append({
571
+ 'revision': rev.group(1) if rev else f.stem,
572
+ 'down_revision': down.group(1).strip() if down else None,
573
+ 'tables_affected': list(set(tables)),
574
+ 'file': str(f.relative_to(PROJECT_ROOT)),
575
+ 'type': 'alembic',
576
+ })
577
+ except Exception:
578
+ pass
579
+
580
+ # SQL files
581
+ for f in sorted(PROJECT_ROOT.rglob('*.sql')):
582
+ if any(p in IGNORE_DIRS for p in f.parts):
583
+ continue
584
+ try:
585
+ source = f.read_text(encoding='utf-8', errors='replace')[:2000]
586
+ tables = re.findall(r'(?:CREATE|ALTER|DROP)\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?(\w+)', source, re.IGNORECASE)
587
+ if tables:
588
+ migrations.append({
589
+ 'revision': f.stem,
590
+ 'tables_affected': list(set(tables)),
591
+ 'file': str(f.relative_to(PROJECT_ROOT)),
592
+ 'type': 'sql',
593
+ })
594
+ except Exception:
595
+ pass
596
+
597
+ return migrations
598
+
599
+
600
+ # ── Frontend Feature Scanner ─────────────────────────────────────────────────
601
+
602
+ class FrontendScanner:
603
+ FEATURE_DIRS = {'features', 'pages', 'views', 'screens', 'modules', 'app'}
604
+ COMPONENT_EXTS = {'.tsx', '.jsx'}
605
+ HOOK_PATTERN = re.compile(r'^use[A-Z]')
606
+
607
+ def scan(self) -> list[dict]:
608
+ features = []
609
+ for feat_dir_name in self.FEATURE_DIRS:
610
+ feat_dir = PROJECT_ROOT / feat_dir_name
611
+ if not feat_dir.exists():
612
+ # try nested src/
613
+ feat_dir = PROJECT_ROOT / 'src' / feat_dir_name
614
+ if not feat_dir.is_dir():
615
+ continue
616
+ for entry in sorted(feat_dir.iterdir()):
617
+ if entry.is_dir() and not entry.name.startswith('.'):
618
+ stats = self._scan_feature_dir(entry)
619
+ stats['name'] = entry.name
620
+ stats['path'] = str(entry.relative_to(PROJECT_ROOT))
621
+ features.append(stats)
622
+ return features
623
+
624
+ def _scan_feature_dir(self, d: Path) -> dict:
625
+ components, hooks, stores, api_files = [], [], [], []
626
+ for f in d.rglob('*'):
627
+ if f.suffix in self.COMPONENT_EXTS:
628
+ if self.HOOK_PATTERN.match(f.stem):
629
+ hooks.append(f.name)
630
+ else:
631
+ components.append(f.name)
632
+ elif 'store' in f.name.lower() or 'slice' in f.name.lower():
633
+ stores.append(f.name)
634
+ elif 'api' in f.name.lower() or 'service' in f.name.lower() or 'client' in f.name.lower():
635
+ api_files.append(f.name)
636
+ return {
637
+ 'components': components,
638
+ 'hooks': hooks,
639
+ 'stores': stores,
640
+ 'api_files': api_files,
641
+ 'component_count': len(components),
642
+ 'hook_count': len(hooks),
643
+ }
644
+
645
+
646
+ # ── Import Chain Tracer ───────────────────────────────────────────────────────
647
+
648
+ class ImportChainTracer:
649
+ MAX_DEPTH = 5
650
+
651
+ def trace(self, routes: list[dict]) -> list[dict]:
652
+ chains = []
653
+ for route in routes[:30]: # limit to first 30 routes
654
+ f = PROJECT_ROOT / route.get('file', '')
655
+ if not f.exists() or f.suffix != '.py':
656
+ continue
657
+ chain = self._trace_file(f, depth=0)
658
+ if len(chain) > 1:
659
+ chains.append({
660
+ 'route': f"{route.get('method','?')} {route.get('path','?')}",
661
+ 'chain': ' → '.join(chain),
662
+ 'file': route.get('file'),
663
+ })
664
+ return chains
665
+
666
+ def _trace_file(self, f: Path, depth: int) -> list[str]:
667
+ if depth > self.MAX_DEPTH or not f.exists():
668
+ return []
669
+ result = [str(f.relative_to(PROJECT_ROOT))]
670
+ try:
671
+ source = f.read_text(encoding='utf-8', errors='replace')
672
+ tree = ast.parse(source)
673
+ for node in ast.walk(tree):
674
+ if isinstance(node, (ast.Import, ast.ImportFrom)):
675
+ if isinstance(node, ast.ImportFrom) and node.module:
676
+ # Attempt to resolve local import
677
+ parts = node.module.split('.')
678
+ candidate = PROJECT_ROOT / Path(*parts).with_suffix('.py')
679
+ if candidate.exists() and depth < self.MAX_DEPTH:
680
+ sub = self._trace_file(candidate, depth + 1)
681
+ if sub:
682
+ result.extend(sub[1:])
683
+ break
684
+ except Exception:
685
+ pass
686
+ return result
687
+
688
+
689
+ # ── Vocabulary Builder ────────────────────────────────────────────────────────
690
+
691
+ class VocabularyBuilder:
692
+ SKIP_WORDS = {'index', 'main', 'base', 'common', 'utils', 'helpers', 'types',
693
+ 'constants', 'config', 'lib', 'api', 'app', 'src', 'test', 'tests'}
694
+
695
+ def build(
696
+ self,
697
+ routes: list[dict],
698
+ models: list[dict],
699
+ schemas: list[dict],
700
+ features: list[dict],
701
+ stack: dict,
702
+ ) -> list[dict]:
703
+ vocab: dict[str, dict] = {}
704
+
705
+ # From features
706
+ for feat in features:
707
+ name = feat['name']
708
+ if name.lower() in self.SKIP_WORDS:
709
+ continue
710
+ aliases = self._name_to_aliases(name)
711
+ for alias in aliases:
712
+ self._add(vocab, alias, 'feature', feat.get('path', ''), f"{feat['component_count']} components")
713
+
714
+ # From models
715
+ for model in models:
716
+ aliases = self._name_to_aliases(model['name'])
717
+ for alias in aliases:
718
+ self._add(vocab, alias, 'model', model.get('file', ''), f"table/model: {model['name']}")
719
+
720
+ # From routes (group by path prefix)
721
+ route_groups: dict[str, list] = {}
722
+ for route in routes:
723
+ prefix = route.get('path', '/').split('/')[1] if '/' in route.get('path', '/') else route.get('path', '')
724
+ if prefix and prefix not in self.SKIP_WORDS:
725
+ route_groups.setdefault(prefix, []).append(route)
726
+ for prefix, group in route_groups.items():
727
+ aliases = self._name_to_aliases(prefix)
728
+ file_ex = group[0].get('file', '')
729
+ for alias in aliases:
730
+ self._add(vocab, alias, 'api', file_ex, f"{len(group)} routes")
731
+
732
+ # Merge learned vocabulary
733
+ learned = self._load_learned()
734
+ for alias, data in learned.items():
735
+ if alias not in vocab and data.get('score', 0) >= 5:
736
+ vocab[alias] = {
737
+ 'alias': alias,
738
+ 'type': 'learned',
739
+ 'location': ', '.join(data.get('targets', [])),
740
+ 'notes': f"learned from session (score: {data['score']:.1f})",
741
+ }
742
+
743
+ return list(vocab.values())
744
+
745
+ def _name_to_aliases(self, name: str) -> list[str]:
746
+ """Generate human aliases from a code name."""
747
+ # camelCase / PascalCase → words
748
+ words = re.sub(r'([A-Z])', r' \1', name).lower().split()
749
+ words = [w for w in words if w and w not in self.SKIP_WORDS]
750
+ if not words:
751
+ return []
752
+ phrase = ' '.join(words)
753
+ aliases = [phrase, name.lower()]
754
+ # e.g. 'deal-pipeline' → 'deal pipeline'
755
+ if '-' in name or '_' in name:
756
+ clean = re.sub(r'[-_]', ' ', name).lower()
757
+ aliases.append(clean)
758
+ return list(dict.fromkeys(aliases)) # dedupe, preserve order
759
+
760
+ def _add(self, vocab: dict, alias: str, type_: str, location: str, notes: str) -> None:
761
+ if alias and alias not in vocab:
762
+ vocab[alias] = {'alias': alias, 'type': type_, 'location': location, 'notes': notes}
763
+
764
+ def _load_learned(self) -> dict:
765
+ if LEARNED_VOC.exists():
766
+ try:
767
+ return json.loads(LEARNED_VOC.read_text())
768
+ except Exception:
769
+ pass
770
+ return {}
771
+
772
+
773
+ # ── Tools & Commands Scanner ──────────────────────────────────────────────────
774
+
775
+ class ToolsScanner:
776
+ def scan(self) -> list[dict]:
777
+ tools = []
778
+
779
+ # npm scripts
780
+ pkg = PROJECT_ROOT / 'package.json'
781
+ if pkg.exists():
782
+ try:
783
+ data = json.loads(pkg.read_text())
784
+ for name, cmd in (data.get('scripts') or {}).items():
785
+ tools.append({'name': name, 'command': f'npm run {name}', 'description': cmd, 'source': 'package.json'})
786
+ except Exception:
787
+ pass
788
+
789
+ # Makefile targets
790
+ makefile = PROJECT_ROOT / 'Makefile'
791
+ if makefile.exists():
792
+ for line in makefile.read_text(encoding='utf-8', errors='replace').splitlines():
793
+ m = re.match(r'^([a-zA-Z][a-zA-Z0-9_-]+)\s*:', line)
794
+ if m and not m.group(1).startswith('.'):
795
+ tools.append({'name': m.group(1), 'command': f'make {m.group(1)}', 'description': '', 'source': 'Makefile'})
796
+
797
+ # Poetry / pip scripts
798
+ for cfg in ['pyproject.toml', 'setup.cfg']:
799
+ path = PROJECT_ROOT / cfg
800
+ if path.exists():
801
+ text = path.read_text(encoding='utf-8', errors='replace')
802
+ for m in re.finditer(r'^\[tool\.poetry\.scripts\]\s*\n((?:\w.*\n)*)', text, re.MULTILINE):
803
+ for line in m.group(1).splitlines():
804
+ if '=' in line:
805
+ name = line.split('=')[0].strip()
806
+ tools.append({'name': name, 'command': name, 'description': '', 'source': cfg})
807
+
808
+ # Shell scripts at root
809
+ for f in PROJECT_ROOT.glob('*.sh'):
810
+ tools.append({'name': f.name, 'command': f'bash {f.name}', 'description': '', 'source': 'shell'})
811
+
812
+ # .claude/skills
813
+ skills_dir = PROJECT_ROOT / '.claude' / 'skills'
814
+ if skills_dir.is_dir():
815
+ for skill_dir in skills_dir.iterdir():
816
+ if (skill_dir / 'SKILL.md').exists():
817
+ tools.append({'name': f'/{skill_dir.name}', 'command': f'/{skill_dir.name}', 'description': 'Claude skill', 'source': 'skills'})
818
+
819
+ return tools
820
+
821
+
822
+ # ── Auth Config Scanner ───────────────────────────────────────────────────────
823
+
824
+ class AuthScanner:
825
+ def scan(self) -> dict:
826
+ info: dict[str, Any] = {'provider': 'unknown', 'files': [], 'patterns': []}
827
+
828
+ patterns_map = {
829
+ 'supabase': ['supabase', 'createClient', 'auth.signIn'],
830
+ 'next-auth': ['NextAuth', 'getSession', 'useSession', 'SessionProvider'],
831
+ 'clerk': ['ClerkProvider', 'useUser', '@clerk'],
832
+ 'auth0': ['Auth0Provider', 'useAuth0', '@auth0'],
833
+ 'jwt': ['jwt.sign', 'jwt.verify', 'create_access_token', 'decode_token'],
834
+ 'passport': ['passport.use', 'passport.authenticate'],
835
+ 'firebase': ['initializeApp', 'getAuth', 'signInWithEmailAndPassword'],
836
+ }
837
+
838
+ for f in PROJECT_ROOT.rglob('*'):
839
+ if any(p in IGNORE_DIRS for p in f.parts):
840
+ continue
841
+ if f.suffix not in ('.py', '.ts', '.tsx', '.js', '.jsx'):
842
+ continue
843
+ try:
844
+ source = f.read_text(encoding='utf-8', errors='replace')
845
+ except Exception:
846
+ continue
847
+ for provider, patterns in patterns_map.items():
848
+ if any(p in source for p in patterns):
849
+ info['provider'] = provider
850
+ rel = str(f.relative_to(PROJECT_ROOT))
851
+ if rel not in info['files']:
852
+ info['files'].append(rel)
853
+ info['patterns'].extend([p for p in patterns if p in source and p not in info['patterns']])
854
+
855
+ return info
856
+
857
+
858
+ # ── Reverse Proxy Scanner ─────────────────────────────────────────────────────
859
+
860
+ class ReverseProxyScanner:
861
+ def scan(self) -> list[dict]:
862
+ rules = []
863
+ # nginx
864
+ for f in PROJECT_ROOT.rglob('*.conf'):
865
+ if any(p in IGNORE_DIRS for p in f.parts):
866
+ continue
867
+ try:
868
+ source = f.read_text(encoding='utf-8', errors='replace')
869
+ for m in re.finditer(r'location\s+([^\s{]+)\s*\{[^}]*proxy_pass\s+([^;]+);', source, re.DOTALL):
870
+ rules.append({'type': 'nginx', 'location': m.group(1).strip(), 'upstream': m.group(2).strip(), 'file': str(f.relative_to(PROJECT_ROOT))})
871
+ except Exception:
872
+ pass
873
+ # Caddyfile
874
+ caddyfile = PROJECT_ROOT / 'Caddyfile'
875
+ if caddyfile.exists():
876
+ try:
877
+ source = caddyfile.read_text(encoding='utf-8', errors='replace')
878
+ for m in re.finditer(r'reverse_proxy\s+([^\n]+)', source):
879
+ rules.append({'type': 'caddy', 'location': '*', 'upstream': m.group(1).strip(), 'file': 'Caddyfile'})
880
+ except Exception:
881
+ pass
882
+ return rules
883
+
884
+
885
+ # ── Dead Code Detector ────────────────────────────────────────────────────────
886
+
887
+ class DeadCodeDetector:
888
+ def detect(self, vocab: list[dict]) -> list[dict]:
889
+ candidates = []
890
+ cutoff_date = '--since=6 months ago'
891
+
892
+ low_score_files = set()
893
+ if LEARNED_VOC.exists():
894
+ try:
895
+ learned = json.loads(LEARNED_VOC.read_text())
896
+ for alias, data in learned.items():
897
+ if data.get('score', 999) < 2:
898
+ low_score_files.update(data.get('targets', []))
899
+ except Exception:
900
+ pass
901
+
902
+ for filepath_str in low_score_files:
903
+ path = PROJECT_ROOT / filepath_str
904
+ if not path.exists():
905
+ continue
906
+ try:
907
+ result = subprocess.run(
908
+ ['git', '-C', str(PROJECT_ROOT), 'log', cutoff_date, '--', filepath_str],
909
+ capture_output=True, text=True, timeout=5
910
+ )
911
+ if not result.stdout.strip():
912
+ candidates.append({
913
+ 'file': filepath_str,
914
+ 'reason': 'No commits in 6 months + low vocabulary score',
915
+ })
916
+ except Exception:
917
+ pass
918
+
919
+ return candidates
920
+
921
+
922
+ # ╔══════════════════════════════════════════════════════════════════════════╗
923
+ # ║ SECTION WRITERS ║
924
+ # ╚══════════════════════════════════════════════════════════════════════════╝
925
+
926
+ def write_section(num: str, name: str, content: str) -> Path:
927
+ filename = f"{num}-{name}.md"
928
+ path = SECTIONS_DIR / filename
929
+ path.write_text(content, encoding='utf-8')
930
+ return path
931
+
932
+ def section_size_kb(path: Path) -> float:
933
+ return path.stat().st_size / 1024 if path.exists() else 0.0
934
+
935
+
936
+ def build_vocabulary_section(vocab: list[dict]) -> str:
937
+ lines = [
938
+ "# Section 01 — Vocabulary Translation Layer\n",
939
+ "> Maps human language to exact code locations. Auto-generated.\n\n",
940
+ "| Alias | Type | Location | Notes |",
941
+ "|-------|------|----------|-------|",
942
+ ]
943
+ for v in sorted(vocab, key=lambda x: x.get('alias', '')):
944
+ alias = v.get('alias', '').replace('|', '\\|')
945
+ type_ = v.get('type', '').replace('|', '\\|')
946
+ location = v.get('location', '').replace('|', '\\|')
947
+ notes = v.get('notes', '').replace('|', '\\|')
948
+ lines.append(f"| {alias} | {type_} | {location} | {notes} |")
949
+ if not vocab:
950
+ lines.append("| _(no vocabulary generated yet — add source code to populate)_ | | | |")
951
+ return '\n'.join(lines) + '\n'
952
+
953
+
954
+ def build_topology_section(services: list[dict]) -> str:
955
+ lines = [
956
+ "# Section 02 — Service Topology\n",
957
+ "> Docker services, ports, and dependencies.\n\n",
958
+ ]
959
+ if not services:
960
+ lines.append("_No docker-compose services detected._\n")
961
+ return '\n'.join(lines)
962
+ lines += ["| Service | Image | Ports | Depends On |",
963
+ "|---------|-------|-------|------------|"]
964
+ for svc in services:
965
+ ports = ', '.join(str(p) for p in svc.get('ports', []))
966
+ deps = ', '.join(svc.get('depends_on', []) if isinstance(svc.get('depends_on'), list)
967
+ else list(svc.get('depends_on', {}).keys()))
968
+ img = str(svc.get('image', '?'))[:50]
969
+ lines.append(f"| {svc['name']} | {img} | {ports} | {deps} |")
970
+ return '\n'.join(lines) + '\n'
971
+
972
+
973
+ def build_environment_section(env_entries: list[dict]) -> str:
974
+ lines = [
975
+ "# Section 03 — Environment Variables\n",
976
+ "> Key names and non-sensitive values only. Secrets are redacted.\n\n",
977
+ ]
978
+ if not env_entries:
979
+ lines.append("_No .env files found._\n")
980
+ return '\n'.join(lines)
981
+ lines += ["| Key | Value | Source |",
982
+ "|-----|-------|--------|"]
983
+ for e in env_entries:
984
+ val = '[SECRET]' if e.get('is_secret') else str(e.get('value', ''))[:60]
985
+ lines.append(f"| `{e['key']}` | `{val}` | {e.get('source', '')} |")
986
+ return '\n'.join(lines) + '\n'
987
+
988
+
989
+ def build_routes_section(routes: list[dict]) -> str:
990
+ lines = [
991
+ "# Section 04 — API Routes\n\n",
992
+ "| Method | Path | File | Line |",
993
+ "|--------|------|------|------|",
994
+ ]
995
+ for r in sorted(routes, key=lambda x: (x.get('path',''), x.get('method',''))):
996
+ lines.append(f"| `{r.get('method','?')}` | `{r.get('path','?')}` | {r.get('file','')} | {r.get('line','')} |")
997
+ if not routes:
998
+ lines.append("| _(no routes detected yet)_ | | | |")
999
+ return '\n'.join(lines) + '\n'
1000
+
1001
+
1002
+ def build_models_section(models: list[dict]) -> str:
1003
+ lines = ["# Section 05 — Data Models\n\n"]
1004
+ if not models:
1005
+ lines.append("_No data models detected yet._\n")
1006
+ return '\n'.join(lines)
1007
+ for model in models:
1008
+ lines.append(f"## {model['name']}")
1009
+ lines.append(f"- **File**: `{model.get('file','?')}`")
1010
+ cols = model.get('columns', [])
1011
+ if cols:
1012
+ lines.append("- **Fields**:")
1013
+ for c in cols[:20]:
1014
+ lines.append(f" - `{c.get('name','?')}` ({c.get('type','?')})")
1015
+ lines.append('')
1016
+ return '\n'.join(lines)
1017
+
1018
+
1019
+ def build_schemas_section(schemas: list[dict]) -> str:
1020
+ lines = ["# Section 06 — Schemas / DTOs\n\n"]
1021
+ if not schemas:
1022
+ lines.append("_No schemas detected yet._\n")
1023
+ return '\n'.join(lines)
1024
+ for s in schemas:
1025
+ lines.append(f"## {s['name']}")
1026
+ lines.append(f"- **File**: `{s.get('file','?')}`")
1027
+ fields = s.get('fields', [])
1028
+ if fields:
1029
+ lines.append(f"- **Fields**: {', '.join(f'`{f}`' for f in fields[:15])}")
1030
+ lines.append('')
1031
+ return '\n'.join(lines)
1032
+
1033
+
1034
+ def build_services_section(routes: list[dict], models: list[dict]) -> str:
1035
+ lines = ["# Section 07 — Services\n\n",
1036
+ "_Service discovery is inferred from directory structure and imports._\n\n"]
1037
+ # Collect unique directories containing routes or models
1038
+ dirs: dict[str, int] = {}
1039
+ for item in routes + models:
1040
+ f = item.get('file', '')
1041
+ d = str(Path(f).parent) if f else ''
1042
+ if d and d != '.':
1043
+ dirs[d] = dirs.get(d, 0) + 1
1044
+ if dirs:
1045
+ lines += ["| Directory | Items |", "|-----------|-------|"]
1046
+ for d, count in sorted(dirs.items(), key=lambda x: -x[1]):
1047
+ lines.append(f"| `{d}` | {count} |")
1048
+ return '\n'.join(lines) + '\n'
1049
+
1050
+
1051
+ def build_background_jobs_section() -> str:
1052
+ lines = ["# Section 08 — Background Jobs\n\n"]
1053
+ jobs = []
1054
+ # Celery
1055
+ for f in PROJECT_ROOT.rglob('*.py'):
1056
+ if any(p in IGNORE_DIRS for p in f.parts):
1057
+ continue
1058
+ try:
1059
+ source = f.read_text(encoding='utf-8', errors='replace')
1060
+ for m in re.finditer(r'@(?:app|celery)\.task|@shared_task', source):
1061
+ # Find function after decorator
1062
+ fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
1063
+ if fn_m:
1064
+ line = source[:m.start()].count('\n') + 1
1065
+ jobs.append({'name': fn_m.group(1), 'type': 'celery', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
1066
+ except Exception:
1067
+ pass
1068
+ # Cron / APScheduler
1069
+ for f in PROJECT_ROOT.rglob('*.py'):
1070
+ if any(p in IGNORE_DIRS for p in f.parts):
1071
+ continue
1072
+ try:
1073
+ source = f.read_text(encoding='utf-8', errors='replace')
1074
+ for m in re.finditer(r'@scheduler\.scheduled_job|scheduler\.add_job|@cron', source):
1075
+ fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
1076
+ if fn_m:
1077
+ line = source[:m.start()].count('\n') + 1
1078
+ jobs.append({'name': fn_m.group(1), 'type': 'scheduler', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
1079
+ except Exception:
1080
+ pass
1081
+
1082
+ if not jobs:
1083
+ lines.append("_No background jobs detected._\n")
1084
+ else:
1085
+ lines += ["| Job | Type | File | Line |", "|-----|------|------|------|"]
1086
+ for j in jobs:
1087
+ lines.append(f"| `{j['name']}` | {j['type']} | {j['file']} | {j.get('line','')} |")
1088
+ return '\n'.join(lines) + '\n'
1089
+
1090
+
1091
+ def build_frontend_section(features: list[dict]) -> str:
1092
+ lines = ["# Section 09 — Frontend Features\n\n"]
1093
+ if not features:
1094
+ lines.append("_No frontend feature directories detected._\n")
1095
+ return '\n'.join(lines)
1096
+ lines += ["| Feature | Components | Hooks | Stores | API Files |",
1097
+ "|---------|-----------|-------|--------|-----------|"]
1098
+ for f in features:
1099
+ lines.append(
1100
+ f"| `{f['path']}` | {f['component_count']} | {f['hook_count']} "
1101
+ f"| {len(f.get('stores',[]))} | {len(f.get('api_files',[]))} |"
1102
+ )
1103
+ return '\n'.join(lines) + '\n'
1104
+
1105
+
1106
+ def build_tools_section(tools: list[dict]) -> str:
1107
+ lines = ["# Section 10 — Tools & Commands\n\n",
1108
+ "| Name | Command | Source |",
1109
+ "|------|---------|--------|"]
1110
+ for t in tools:
1111
+ lines.append(f"| `{t['name']}` | `{t['command']}` | {t.get('source','')} |")
1112
+ if not tools:
1113
+ lines.append("| _(no tools detected)_ | | |")
1114
+ return '\n'.join(lines) + '\n'
1115
+
1116
+
1117
+ def build_migrations_section(migrations: list[dict]) -> str:
1118
+ lines = ["# Section 11 — Migrations\n\n"]
1119
+ if not migrations:
1120
+ lines.append("_No migrations detected._\n")
1121
+ return '\n'.join(lines)
1122
+ lines += ["| Revision | Tables Affected | Type | File |",
1123
+ "|----------|----------------|------|------|"]
1124
+ for m in migrations[-30:]: # last 30
1125
+ tables = ', '.join(m.get('tables_affected', []))[:60]
1126
+ lines.append(f"| `{m.get('revision','?')}` | {tables} | {m.get('type','?')} | {m.get('file','')} |")
1127
+ return '\n'.join(lines) + '\n'
1128
+
1129
+
1130
+ def build_import_chains_section(chains: list[dict]) -> str:
1131
+ lines = ["# Section 12 — Import Chains\n\n",
1132
+ "_Traces route → service → model → table for key endpoints._\n\n"]
1133
+ if not chains:
1134
+ lines.append("_No import chains traced (requires Python source files with routes)._\n")
1135
+ return '\n'.join(lines)
1136
+ for c in chains:
1137
+ lines.append(f"**{c['route']}**")
1138
+ lines.append(f"```\n{c['chain']}\n```\n")
1139
+ return '\n'.join(lines)
1140
+
1141
+
1142
+ def build_frontend_backend_section(routes: list[dict], features: list[dict]) -> str:
1143
+ lines = ["# Section 13 — Frontend → Backend Map\n\n",
1144
+ "_Maps frontend API service calls to backend route paths._\n\n"]
1145
+ mappings = []
1146
+ api_calls: list[tuple[str,str]] = []
1147
+
1148
+ for feat in features:
1149
+ for api_file in feat.get('api_files', []):
1150
+ full = PROJECT_ROOT / feat['path'] / api_file
1151
+ if full.exists():
1152
+ try:
1153
+ source = full.read_text(encoding='utf-8', errors='replace')
1154
+ for m in re.finditer(r'[\'"`](/api/[^\'"` \n]+)', source):
1155
+ api_calls.append((m.group(1), str(full.relative_to(PROJECT_ROOT))))
1156
+ except Exception:
1157
+ pass
1158
+
1159
+ route_paths = {r.get('path',''): r for r in routes}
1160
+ for call, fe_file in api_calls:
1161
+ if call in route_paths:
1162
+ be = route_paths[call]
1163
+ mappings.append({'frontend': fe_file, 'url': call, 'backend': be.get('file','?')})
1164
+
1165
+ if not mappings:
1166
+ lines.append("_No frontend→backend mappings detected yet._\n")
1167
+ else:
1168
+ lines += ["| Frontend File | URL | Backend File |",
1169
+ "|--------------|-----|--------------|"]
1170
+ for m in mappings:
1171
+ lines.append(f"| {m['frontend']} | `{m['url']}` | {m['backend']} |")
1172
+ return '\n'.join(lines) + '\n'
1173
+
1174
+
1175
+ def build_proxy_section(proxy_rules: list[dict]) -> str:
1176
+ lines = ["# Section 14 — Reverse Proxy\n\n"]
1177
+ if not proxy_rules:
1178
+ lines.append("_No reverse proxy configuration detected._\n")
1179
+ return '\n'.join(lines)
1180
+ lines += ["| Type | Location | Upstream | File |",
1181
+ "|------|----------|----------|------|"]
1182
+ for r in proxy_rules:
1183
+ lines.append(f"| {r['type']} | `{r['location']}` | `{r['upstream']}` | {r['file']} |")
1184
+ return '\n'.join(lines) + '\n'
1185
+
1186
+
1187
+ def build_auth_section(auth_info: dict) -> str:
1188
+ lines = ["# Section 15 — Auth Configuration\n\n",
1189
+ f"**Provider**: {auth_info.get('provider','unknown')}\n\n"]
1190
+ files = auth_info.get('files', [])
1191
+ if files:
1192
+ lines.append("**Auth files:**")
1193
+ for f in files[:10]:
1194
+ lines.append(f"- `{f}`")
1195
+ patterns = auth_info.get('patterns', [])
1196
+ if patterns:
1197
+ lines.append(f"\n**Detected patterns**: {', '.join(f'`{p}`' for p in patterns[:10])}")
1198
+ return '\n'.join(lines) + '\n'
1199
+
1200
+
1201
+ def build_infra_section(stack: dict, services: list[dict]) -> str:
1202
+ lines = ["# Section 16 — Infrastructure Profile\n\n",
1203
+ f"| Property | Value |",
1204
+ "|----------|-------|",
1205
+ f"| **Language** | {stack.get('language','?')} |",
1206
+ f"| **Framework** | {stack.get('framework','?')} |",
1207
+ f"| **Database** | {stack.get('database','?')} |",
1208
+ f"| **ORM** | {stack.get('orm','?')} |",
1209
+ f"| **Auth** | {stack.get('auth','?')} |",
1210
+ f"| **Package Manager** | {stack.get('package_manager','?')} |",
1211
+ f"| **Infrastructure** | {stack.get('infrastructure','?')} |",
1212
+ f"| **Services** | {len(services)} docker services |",
1213
+ ""]
1214
+ return '\n'.join(lines)
1215
+
1216
+
1217
+ def build_learned_vocab_section() -> str:
1218
+ lines = ["# Section 17 — Learned Vocabulary\n\n",
1219
+ "_Aliases mined from Claude Code session history. Score = frequency × recency._\n\n"]
1220
+ learned = {}
1221
+ if LEARNED_VOC.exists():
1222
+ try:
1223
+ learned = json.loads(LEARNED_VOC.read_text())
1224
+ except Exception:
1225
+ pass
1226
+ if not learned:
1227
+ lines.append("_No session-mined vocabulary yet. Accumulates over time._\n")
1228
+ return '\n'.join(lines)
1229
+ lines += ["| Alias | Score | Targets | Last Seen |",
1230
+ "|-------|-------|---------|-----------|"]
1231
+ for alias, data in sorted(learned.items(), key=lambda x: -x[1].get('score', 0)):
1232
+ if data.get('score', 0) >= 2:
1233
+ targets = ', '.join(data.get('targets', []))[:60]
1234
+ lines.append(f"| {alias} | {data.get('score',0):.1f} | {targets} | {data.get('last_seen','?')} |")
1235
+ return '\n'.join(lines) + '\n'
1236
+
1237
+
1238
+ def build_dead_code_section(candidates: list[dict]) -> str:
1239
+ lines = ["# Section 18 — Dead Code Candidates\n\n",
1240
+ "> Files flagged for human review: no recent commits AND low vocabulary score.\n",
1241
+ "> **Do not auto-delete.** Review before removing.\n\n"]
1242
+ if not candidates:
1243
+ lines.append("_No dead code candidates detected._\n")
1244
+ return '\n'.join(lines)
1245
+ for c in candidates:
1246
+ lines.append(f"- `{c['file']}` — {c.get('reason','')}")
1247
+ return '\n'.join(lines) + '\n'
1248
+
1249
+
1250
+ def build_doc_pointers_section() -> str:
1251
+ lines = ["# Section 19 — Documentation Pointers\n\n"]
1252
+ docs = []
1253
+ doc_dirs = ['docs', 'doc', 'documentation', 'wiki', '.docs']
1254
+ doc_exts = {'.md', '.rst', '.txt', '.adoc'}
1255
+
1256
+ for doc_dir in doc_dirs:
1257
+ d = PROJECT_ROOT / doc_dir
1258
+ if d.is_dir():
1259
+ for f in sorted(d.rglob('*')):
1260
+ if f.is_file() and f.suffix in doc_exts:
1261
+ docs.append(str(f.relative_to(PROJECT_ROOT)))
1262
+
1263
+ # Root-level docs
1264
+ for f in PROJECT_ROOT.glob('*.md'):
1265
+ docs.append(str(f.relative_to(PROJECT_ROOT)))
1266
+
1267
+ if not docs:
1268
+ lines.append("_No documentation files found._\n")
1269
+ else:
1270
+ for d in docs[:30]:
1271
+ lines.append(f"- [`{d}`]({d})")
1272
+ return '\n'.join(lines) + '\n'
1273
+
1274
+
1275
+ # ╔══════════════════════════════════════════════════════════════════════════╗
1276
+ # ║ PROJECT MAP TOC ║
1277
+ # ╚══════════════════════════════════════════════════════════════════════════╝
1278
+
1279
+ def build_project_map(
1280
+ stack: dict,
1281
+ routes: list[dict],
1282
+ models: list[dict],
1283
+ schemas: list[dict],
1284
+ features: list[dict],
1285
+ migrations: list[dict],
1286
+ services: list[dict],
1287
+ vocab: list[dict],
1288
+ section_files: list[tuple[str, Path]],
1289
+ ) -> str:
1290
+ now = datetime.now().strftime('%Y-%m-%d %H:%M')
1291
+ name = stack.get('name', PROJECT_ROOT.name)
1292
+ slug = stack.get('slug', name)
1293
+
1294
+ lines = [
1295
+ f"# {name} — Project Map\n",
1296
+ f"> Auto-generated by `generate.py` on {now}. Do not edit manually.\n\n",
1297
+ f"## Stats\n",
1298
+ f"| Metric | Count |",
1299
+ f"|--------|-------|",
1300
+ f"| API Routes | {len(routes)} |",
1301
+ f"| Data Models | {len(models)} |",
1302
+ f"| Schemas/DTOs | {len(schemas)} |",
1303
+ f"| Frontend Features | {len(features)} |",
1304
+ f"| Migrations | {len(migrations)} |",
1305
+ f"| Docker Services | {len(services)} |",
1306
+ f"| Vocabulary Entries | {len(vocab)} |",
1307
+ f"| Stack | {stack.get('language','?')} / {stack.get('framework','?')} |",
1308
+ "",
1309
+ "## Section Index\n",
1310
+ "| # | Section | Size | When to Read |",
1311
+ "|---|---------|------|--------------|",
1312
+ ]
1313
+
1314
+ WHEN_TO_READ = {
1315
+ '01': 'Any task — start here if you don\'t know where the code lives',
1316
+ '02': 'Debugging connectivity, adding a service, understanding ports',
1317
+ '03': 'Environment setup, missing vars, config issues',
1318
+ '04': 'Adding/editing API endpoints, checking what routes exist',
1319
+ '05': 'Changing database schema, adding fields, understanding relations',
1320
+ '06': 'Adding DTOs, changing request/response shapes',
1321
+ '07': 'Adding service logic, understanding service boundaries',
1322
+ '08': 'Working with background jobs, queues, scheduled tasks',
1323
+ '09': 'Frontend feature work, understanding UI structure',
1324
+ '10': 'Available commands, scripts, developer tooling',
1325
+ '11': 'Database migrations, schema history',
1326
+ '12': 'Tracing data flow from HTTP request to DB',
1327
+ '13': 'Understanding which frontend calls which backend endpoint',
1328
+ '14': 'Proxy routing, nginx/caddy config',
1329
+ '15': 'Auth flow, sessions, permissions',
1330
+ '16': 'Infrastructure overview, tech stack summary',
1331
+ '17': 'Vocabulary learned from past sessions',
1332
+ '18': 'Dead code review',
1333
+ '19': 'Finding documentation, READMEs, wikis',
1334
+ }
1335
+
1336
+ for name_part, path in section_files:
1337
+ num = name_part.split('-')[0]
1338
+ display = name_part.replace('-', ' ').title()
1339
+ size_kb = section_size_kb(path)
1340
+ when = WHEN_TO_READ.get(num, '')
1341
+ lines.append(f"| [{num}](sections/{path.name}) | {display} | {size_kb:.1f} KB | {when} |")
1342
+
1343
+ lines += [
1344
+ "",
1345
+ "## Quick Routing\n",
1346
+ "| Task | Read Sections |",
1347
+ "|------|--------------|",
1348
+ "| Feature / UX work | 01 → 09 → 04 |",
1349
+ "| Add model or field | 05 → 06 → 12 |",
1350
+ "| Troubleshoot error | 02 → 03 → 14 |",
1351
+ "| Infrastructure / scaling | 16 → 02 |",
1352
+ "| Auth / security | 15 → 19 |",
1353
+ "| What tools exist | 10 |",
1354
+ "| Background jobs | 08 |",
1355
+ "| Migration history | 11 |",
1356
+ "",
1357
+ f"## Regenerate\n",
1358
+ "```bash",
1359
+ "python .claude/project-map/generate.py # skip if unchanged",
1360
+ "python .claude/project-map/generate.py --force # always regenerate",
1361
+ "```",
1362
+ ]
1363
+ return '\n'.join(lines) + '\n'
1364
+
1365
+
1366
+ # ╔══════════════════════════════════════════════════════════════════════════╗
1367
+ # ║ MAIN ║
1368
+ # ╚══════════════════════════════════════════════════════════════════════════╝
1369
+
1370
+ def main() -> None:
1371
+ parser = argparse.ArgumentParser(description='Babel Fish — generate project map')
1372
+ parser.add_argument('--force', action='store_true', help='Force regeneration even if checksums match')
1373
+ parser.add_argument('--project-root', type=Path, default=None, help='Override project root')
1374
+ parser.add_argument('--stack-json', type=Path, default=None, help='Path to stack.json from detect-stack.sh')
1375
+ args = parser.parse_args()
1376
+
1377
+ global PROJECT_ROOT
1378
+ if args.project_root:
1379
+ PROJECT_ROOT = args.project_root.resolve()
1380
+
1381
+ print(f"[generate] Project root: {PROJECT_ROOT}")
1382
+
1383
+ # 1. Checksum check
1384
+ watched = collect_watched_files()
1385
+ checksum = compute_checksum(watched)
1386
+
1387
+ if not args.force and is_unchanged(checksum):
1388
+ print("[generate] ✓ No changes detected — skipping regeneration (use --force to override)")
1389
+ sys.exit(0)
1390
+
1391
+ # 2. Load stack
1392
+ stack = load_stack(args.stack_json)
1393
+ print(f"[generate] Stack: {stack['language']} / {stack['framework']}")
1394
+
1395
+ # 3. Collect all relevant files
1396
+ py_files = [f for f in watched if f.suffix == '.py']
1397
+ ts_files = [f for f in watched if f.suffix in ('.ts','.tsx','.js','.jsx')]
1398
+ go_files = [f for f in watched if f.suffix == '.go']
1399
+
1400
+ # 4. Parse
1401
+ print("[generate] Parsing routes...")
1402
+ routes: list[dict] = []
1403
+ if stack['language'] in ('python', 'unknown'):
1404
+ routes.extend(PythonRouteParser().parse(py_files))
1405
+ if stack['language'] in ('typescript', 'javascript', 'unknown'):
1406
+ routes.extend(TSRouteParser().parse(ts_files))
1407
+ if stack['language'] in ('go', 'unknown'):
1408
+ routes.extend(GoRouteParser().parse(go_files))
1409
+
1410
+ print("[generate] Parsing models...")
1411
+ models: list[dict] = []
1412
+ if stack['language'] in ('python', 'unknown'):
1413
+ models.extend(PythonModelParser().parse(py_files))
1414
+ if stack['language'] in ('typescript', 'javascript', 'unknown'):
1415
+ models.extend(TSModelParser().parse(ts_files + [f for f in watched if f.suffix == '.prisma']))
1416
+ if stack['language'] in ('go', 'unknown'):
1417
+ models.extend(GoModelParser().parse(go_files))
1418
+
1419
+ print("[generate] Parsing schemas...")
1420
+ schemas = PythonSchemaParser().parse(py_files)
1421
+
1422
+ print("[generate] Scanning environment...")
1423
+ services = DockerComposeParser().parse()
1424
+ env_entries = EnvParser().parse()
1425
+ migrations = MigrationParser().parse()
1426
+ features = FrontendScanner().scan()
1427
+ tools = ToolsScanner().scan()
1428
+ auth_info = AuthScanner().scan()
1429
+ proxy_rules = ReverseProxyScanner().scan()
1430
+
1431
+ print("[generate] Building vocabulary...")
1432
+ vocab = VocabularyBuilder().build(routes, models, schemas, features, stack)
1433
+
1434
+ print("[generate] Tracing import chains...")
1435
+ chains = ImportChainTracer().trace(routes)
1436
+
1437
+ print("[generate] Detecting dead code candidates...")
1438
+ dead_code = DeadCodeDetector().detect(vocab)
1439
+
1440
+ # 5. Write sections
1441
+ print("[generate] Writing sections...")
1442
+ section_files: list[tuple[str, Path]] = [
1443
+ ('01-vocabulary', write_section('01', 'vocabulary', build_vocabulary_section(vocab))),
1444
+ ('02-service-topology', write_section('02', 'service-topology', build_topology_section(services))),
1445
+ ('03-environment', write_section('03', 'environment', build_environment_section(env_entries))),
1446
+ ('04-api-routes', write_section('04', 'api-routes', build_routes_section(routes))),
1447
+ ('05-data-models', write_section('05', 'data-models', build_models_section(models))),
1448
+ ('06-schemas', write_section('06', 'schemas', build_schemas_section(schemas))),
1449
+ ('07-services', write_section('07', 'services', build_services_section(routes, models))),
1450
+ ('08-background-jobs', write_section('08', 'background-jobs', build_background_jobs_section())),
1451
+ ('09-frontend-features', write_section('09', 'frontend-features', build_frontend_section(features))),
1452
+ ('10-tools-commands', write_section('10', 'tools-commands', build_tools_section(tools))),
1453
+ ('11-migrations', write_section('11', 'migrations', build_migrations_section(migrations))),
1454
+ ('12-import-chains', write_section('12', 'import-chains', build_import_chains_section(chains))),
1455
+ ('13-frontend-backend-map',write_section('13', 'frontend-backend-map',build_frontend_backend_section(routes, features))),
1456
+ ('14-reverse-proxy', write_section('14', 'reverse-proxy', build_proxy_section(proxy_rules))),
1457
+ ('15-auth-config', write_section('15', 'auth-config', build_auth_section(auth_info))),
1458
+ ('16-infra-profile', write_section('16', 'infra-profile', build_infra_section(stack, services))),
1459
+ ('17-learned-vocabulary', write_section('17', 'learned-vocabulary', build_learned_vocab_section())),
1460
+ ('18-dead-code', write_section('18', 'dead-code', build_dead_code_section(dead_code))),
1461
+ ('19-doc-pointers', write_section('19', 'doc-pointers', build_doc_pointers_section())),
1462
+ ]
1463
+
1464
+ # 6. Write PROJECT_MAP.md
1465
+ print("[generate] Writing PROJECT_MAP.md...")
1466
+ project_map = build_project_map(stack, routes, models, schemas, features, migrations, services, vocab, section_files)
1467
+ (MAP_DIR / 'PROJECT_MAP.md').write_text(project_map, encoding='utf-8')
1468
+
1469
+ # 7. Update checksums
1470
+ save_checksums({'input_hash': checksum, 'generated_at': datetime.now().isoformat(), 'route_count': len(routes), 'model_count': len(models)})
1471
+
1472
+ # 8. Ensure learned-vocabulary.json exists
1473
+ if not LEARNED_VOC.exists():
1474
+ LEARNED_VOC.write_text('{}', encoding='utf-8')
1475
+
1476
+ total_kb = sum(section_size_kb(p) for _, p in section_files)
1477
+ print(f"[generate] ✓ Done — {len(routes)} routes, {len(models)} models, {len(vocab)} vocab entries, {total_kb:.1f} KB total")
1478
+
1479
+
1480
+ if __name__ == '__main__':
1481
+ main()