@theglitchking/babel-fish 1.0.2 → 2.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.claude/install.sh +665 -0
  2. package/.claude/project-map/PROJECT_MAP.md +61 -0
  3. package/.claude/project-map/__pycache__/generate.cpython-312.pyc +0 -0
  4. package/.claude/project-map/checksums.json +6 -0
  5. package/.claude/project-map/generate.py +1494 -0
  6. package/.claude/project-map/grader.py +583 -0
  7. package/.claude/project-map/learned-vocabulary.json +1 -0
  8. package/.claude/project-map/mine-sessions.py +419 -0
  9. package/.claude/project-map/reports/install-report.md +54 -0
  10. package/.claude/project-map/reports/iteration-01-report.md +54 -0
  11. package/.claude/project-map/reports/iteration-01-score.json +7 -0
  12. package/.claude/project-map/sections/01-vocabulary.md +8 -0
  13. package/.claude/project-map/sections/02-service-topology.md +6 -0
  14. package/.claude/project-map/sections/03-environment.md +6 -0
  15. package/.claude/project-map/sections/04-api-routes.md +6 -0
  16. package/.claude/project-map/sections/05-data-models.md +4 -0
  17. package/.claude/project-map/sections/06-schemas.md +4 -0
  18. package/.claude/project-map/sections/07-services.md +6 -0
  19. package/.claude/project-map/sections/08-background-jobs.md +5 -0
  20. package/.claude/project-map/sections/09-frontend-features.md +4 -0
  21. package/.claude/project-map/sections/10-tools-commands.md +8 -0
  22. package/.claude/project-map/sections/11-migrations.md +4 -0
  23. package/.claude/project-map/sections/12-import-chains.md +7 -0
  24. package/.claude/project-map/sections/13-frontend-backend-map.md +8 -0
  25. package/.claude/project-map/sections/14-reverse-proxy.md +4 -0
  26. package/.claude/project-map/sections/15-auth-config.md +6 -0
  27. package/.claude/project-map/sections/16-infra-profile.md +13 -0
  28. package/.claude/project-map/sections/17-learned-vocabulary.md +7 -0
  29. package/.claude/project-map/sections/18-dead-code.md +9 -0
  30. package/.claude/project-map/sections/19-doc-pointers.md +5 -0
  31. package/.claude/project-map/stack.json +12 -0
  32. package/.claude/rules/operational-runbook.md +40 -0
  33. package/.claude/rules/project-vocabulary.md +25 -0
  34. package/.claude/scripts/detect-stack.sh +222 -0
  35. package/.claude/scripts/ensure-python.sh +100 -0
  36. package/.claude/scripts/statusline.sh +27 -0
  37. package/.claude/scripts/validate.sh +59 -0
  38. package/.claude/settings.json +6 -0
  39. package/.claude/settings.local.json +6 -0
  40. package/.claude/skills/babel-fish-developer-skill/SKILL.md +56 -0
  41. package/.claude/templates/SKILL.md.template +56 -0
  42. package/.claude/templates/operational-runbook.md.template +40 -0
  43. package/.claude/templates/project-vocabulary.md.template +25 -0
  44. package/.claude-plugin/marketplace.json +2 -2
  45. package/.claude-plugin/plugin.json +1 -1
  46. package/.githooks/install.sh +4 -0
  47. package/.githooks/pre-commit +22 -0
  48. package/CHANGELOG.md +88 -0
  49. package/README.md +21 -0
  50. package/bin/babel-fish.js +78 -71
  51. package/commands/policy.md +16 -0
  52. package/commands/relink.md +6 -0
  53. package/commands/status.md +6 -0
  54. package/commands/update.md +6 -0
  55. package/hooks/hooks.json +15 -0
  56. package/hooks/session-start.js +11 -0
  57. package/package.json +19 -3
  58. package/scripts/link-skills.js +31 -0
@@ -0,0 +1,1494 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ generate.py — Babel Fish
4
+ Stack-agnostic introspection script for the babel-fish codebase mapper plugin.
5
+ Produces a split-section project map under .claude/project-map/sections/
6
+
7
+ Usage:
8
+ python generate.py [--force] [--project-root PATH] [--stack-json PATH]
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import ast
13
+ import argparse
14
+ import hashlib
15
+ import json
16
+ import os
17
+ import re
18
+ import subprocess
19
+ import sys
20
+ from datetime import datetime
21
+ from pathlib import Path
22
+ from typing import Any
23
+
24
+ # ── Try optional deps ────────────────────────────────────────────────────────
25
+ try:
26
+ import yaml
27
+ HAS_YAML = True
28
+ except ImportError:
29
+ HAS_YAML = False
30
+
31
+ # ── Paths ────────────────────────────────────────────────────────────────────
32
+ SCRIPT_DIR = Path(__file__).parent
33
+ MAP_DIR = SCRIPT_DIR
34
+ SECTIONS_DIR = MAP_DIR / "sections"
35
+ CHECKSUMS = MAP_DIR / "checksums.json"
36
+ LEARNED_VOC = MAP_DIR / "learned-vocabulary.json"
37
+
38
+ # Resolve project root: two levels up from .claude/project-map/
39
+ PROJECT_ROOT = MAP_DIR.parent.parent
40
+
41
+ SECTIONS_DIR.mkdir(parents=True, exist_ok=True)
42
+
43
+ # ── Secrets guard ────────────────────────────────────────────────────────────
44
+ SECRET_PATTERNS = re.compile(
45
+ r'(?i)(password|secret|token|api_key|apikey|private_key|auth_token|'
46
+ r'access_key|secret_key|client_secret|db_pass|database_password|'
47
+ r'stripe_key|twilio_auth|sendgrid_key|aws_secret)\s*[=:]\s*\S+'
48
+ )
49
+
50
+ def redact_secrets(text: str) -> str:
51
+ return SECRET_PATTERNS.sub(r'[REDACTED]', text)
52
+
53
+
54
+ # ── Checksum logic ───────────────────────────────────────────────────────────
55
+ WATCHED_EXTENSIONS = {
56
+ '.py', '.ts', '.tsx', '.js', '.jsx', '.go', '.java', '.kt',
57
+ '.yaml', '.yml', '.toml', '.json', '.prisma', '.sql', '.env',
58
+ }
59
+ WATCHED_NAMES = {
60
+ 'docker-compose.yml', 'docker-compose.yaml', 'docker-compose.dev.yml',
61
+ 'package.json', 'requirements.txt', 'pyproject.toml', 'go.mod',
62
+ 'Cargo.toml', 'pom.xml', 'Gemfile', 'Makefile',
63
+ }
64
+ IGNORE_DIRS = {
65
+ '.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env',
66
+ 'dist', 'build', '.next', '.nuxt', 'target', 'vendor', '.cache',
67
+ '.claude', '.planning',
68
+ }
69
+
70
+ def collect_watched_files() -> list[Path]:
71
+ result = []
72
+ for path in sorted(PROJECT_ROOT.rglob('*')):
73
+ if any(p in IGNORE_DIRS for p in path.parts):
74
+ continue
75
+ if path.is_file() and (path.suffix in WATCHED_EXTENSIONS or path.name in WATCHED_NAMES):
76
+ result.append(path)
77
+ return result
78
+
79
+ def compute_checksum(files: list[Path]) -> str:
80
+ h = hashlib.sha256()
81
+ for f in files:
82
+ h.update(str(f).encode())
83
+ try:
84
+ h.update(str(f.stat().st_mtime_ns).encode())
85
+ except OSError:
86
+ pass
87
+ return h.hexdigest()
88
+
89
+ def load_checksums() -> dict:
90
+ if CHECKSUMS.exists():
91
+ try:
92
+ return json.loads(CHECKSUMS.read_text())
93
+ except Exception:
94
+ pass
95
+ return {}
96
+
97
+ def save_checksums(data: dict) -> None:
98
+ CHECKSUMS.write_text(json.dumps(data, indent=2))
99
+
100
+ def is_unchanged(checksum: str) -> bool:
101
+ stored = load_checksums()
102
+ return stored.get('input_hash') == checksum
103
+
104
+
105
+ # ── Stack detection (reads stack.json if present, else fallback) ─────────────
106
+ def load_stack(stack_json_path: Path | None = None) -> dict:
107
+ candidates = [
108
+ stack_json_path,
109
+ MAP_DIR / "stack.json",
110
+ PROJECT_ROOT / ".claude" / "stack.json",
111
+ ]
112
+ for path in candidates:
113
+ if path and path.exists():
114
+ try:
115
+ return json.loads(path.read_text())
116
+ except Exception:
117
+ pass
118
+ return {
119
+ "name": PROJECT_ROOT.name,
120
+ "slug": re.sub(r'[^a-z0-9]', '-', PROJECT_ROOT.name.lower()).strip('-'),
121
+ "language": "unknown",
122
+ "framework": "unknown",
123
+ "database": "unknown",
124
+ "orm": "unknown",
125
+ "auth": "unknown",
126
+ "package_manager": "unknown",
127
+ "infrastructure": "none",
128
+ "project_root": str(PROJECT_ROOT),
129
+ }
130
+
131
+
132
+ # ╔══════════════════════════════════════════════════════════════════════════╗
133
+ # ║ PARSERS ║
134
+ # ╚══════════════════════════════════════════════════════════════════════════╝
135
+
136
+ # ── Python / FastAPI / Django / Flask ────────────────────────────────────────
137
+
138
+ class PythonRouteParser:
139
+ """Extract routes from Python routers using ast."""
140
+
141
+ DECORATOR_PATTERNS = re.compile(
142
+ r'@(router|app|api_router|blueprint)\.(get|post|put|patch|delete|head|options|websocket)\s*\('
143
+ )
144
+
145
+ def parse(self, files: list[Path]) -> list[dict]:
146
+ routes = []
147
+ for f in files:
148
+ if f.suffix != '.py':
149
+ continue
150
+ try:
151
+ source = f.read_text(encoding='utf-8', errors='replace')
152
+ tree = ast.parse(source, filename=str(f))
153
+ routes.extend(self._extract_routes(tree, f, source))
154
+ except SyntaxError:
155
+ pass
156
+ return routes
157
+
158
+ def _extract_routes(self, tree: ast.AST, filepath: Path, source: str) -> list[dict]:
159
+ routes = []
160
+ lines = source.splitlines()
161
+ for node in ast.walk(tree):
162
+ if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
163
+ continue
164
+ for decorator in node.decorator_list:
165
+ info = self._parse_decorator(decorator, lines, node.lineno)
166
+ if info:
167
+ info['file'] = str(filepath.relative_to(PROJECT_ROOT))
168
+ info['line'] = node.lineno
169
+ info['function'] = node.name
170
+ routes.append(info)
171
+ return routes
172
+
173
+ def _parse_decorator(self, decorator: ast.expr, lines: list[str], lineno: int) -> dict | None:
174
+ if isinstance(decorator, ast.Call):
175
+ func = decorator.func
176
+ if isinstance(func, ast.Attribute) and func.attr in (
177
+ 'get','post','put','patch','delete','head','options','websocket'
178
+ ):
179
+ method = func.attr.upper()
180
+ path = ''
181
+ if decorator.args:
182
+ arg = decorator.args[0]
183
+ if isinstance(arg, ast.Constant):
184
+ path = str(arg.value)
185
+ return {'method': method, 'path': path}
186
+ return None
187
+
188
+
189
+ class PythonModelParser:
190
+ """Extract SQLAlchemy / Django ORM models using ast."""
191
+
192
+ def parse(self, files: list[Path]) -> list[dict]:
193
+ models = []
194
+ for f in files:
195
+ if f.suffix != '.py':
196
+ continue
197
+ try:
198
+ source = f.read_text(encoding='utf-8', errors='replace')
199
+ tree = ast.parse(source, filename=str(f))
200
+ models.extend(self._extract_models(tree, f))
201
+ except SyntaxError:
202
+ pass
203
+ return models
204
+
205
+ def _extract_models(self, tree: ast.AST, filepath: Path) -> list[dict]:
206
+ models = []
207
+ for node in ast.walk(tree):
208
+ if not isinstance(node, ast.ClassDef):
209
+ continue
210
+ bases = [self._base_name(b) for b in node.bases]
211
+ is_model = any(
212
+ b in ('Base', 'Model', 'BaseModel', 'DeclarativeBase', 'AbstractModel')
213
+ or 'Model' in b
214
+ for b in bases if b
215
+ )
216
+ if not is_model:
217
+ continue
218
+ columns = self._extract_columns(node)
219
+ models.append({
220
+ 'name': node.name,
221
+ 'file': str(filepath.relative_to(PROJECT_ROOT)),
222
+ 'line': node.lineno,
223
+ 'bases': bases,
224
+ 'columns': columns,
225
+ })
226
+ return models
227
+
228
+ def _base_name(self, node: ast.expr) -> str:
229
+ if isinstance(node, ast.Name):
230
+ return node.id
231
+ if isinstance(node, ast.Attribute):
232
+ return node.attr
233
+ return ''
234
+
235
+ def _extract_columns(self, class_node: ast.ClassDef) -> list[dict]:
236
+ cols = []
237
+ for node in ast.walk(class_node):
238
+ if isinstance(node, ast.AnnAssign) and isinstance(node.target, ast.Name):
239
+ cols.append({
240
+ 'name': node.target.id,
241
+ 'type': ast.unparse(node.annotation) if hasattr(ast, 'unparse') else '?',
242
+ })
243
+ elif isinstance(node, ast.Assign):
244
+ for target in node.targets:
245
+ if isinstance(target, ast.Name):
246
+ if isinstance(node.value, ast.Call):
247
+ func_name = ''
248
+ if isinstance(node.value.func, ast.Name):
249
+ func_name = node.value.func.id
250
+ elif isinstance(node.value.func, ast.Attribute):
251
+ func_name = node.value.func.attr
252
+ if func_name in ('Column', 'Field', 'CharField', 'IntegerField',
253
+ 'TextField', 'BooleanField', 'ForeignKey',
254
+ 'ManyToManyField', 'DateTimeField', 'mapped_column'):
255
+ cols.append({'name': target.id, 'type': func_name})
256
+ return cols
257
+
258
+
259
+ class PythonSchemaParser:
260
+ """Extract Pydantic schemas / Django serializers using ast."""
261
+
262
+ def parse(self, files: list[Path]) -> list[dict]:
263
+ schemas = []
264
+ for f in files:
265
+ if f.suffix != '.py':
266
+ continue
267
+ try:
268
+ source = f.read_text(encoding='utf-8', errors='replace')
269
+ tree = ast.parse(source, filename=str(f))
270
+ schemas.extend(self._extract_schemas(tree, f))
271
+ except SyntaxError:
272
+ pass
273
+ return schemas
274
+
275
+ def _extract_schemas(self, tree: ast.AST, filepath: Path) -> list[dict]:
276
+ schemas = []
277
+ for node in ast.walk(tree):
278
+ if not isinstance(node, ast.ClassDef):
279
+ continue
280
+ bases = [self._base_name(b) for b in node.bases]
281
+ is_schema = any(
282
+ b in ('BaseModel', 'Schema', 'Serializer', 'ModelSerializer',
283
+ 'TypedDict', 'NamedTuple')
284
+ or 'Schema' in b or 'Serializer' in b
285
+ for b in bases if b
286
+ )
287
+ if not is_schema:
288
+ continue
289
+ fields = [
290
+ n.target.id
291
+ for n in ast.walk(node)
292
+ if isinstance(n, ast.AnnAssign) and isinstance(n.target, ast.Name)
293
+ ]
294
+ schemas.append({
295
+ 'name': node.name,
296
+ 'file': str(filepath.relative_to(PROJECT_ROOT)),
297
+ 'line': node.lineno,
298
+ 'fields': fields,
299
+ })
300
+ return schemas
301
+
302
+ def _base_name(self, node: ast.expr) -> str:
303
+ if isinstance(node, ast.Name):
304
+ return node.id
305
+ if isinstance(node, ast.Attribute):
306
+ return node.attr
307
+ return ''
308
+
309
+
310
+ # ── TypeScript / JavaScript ──────────────────────────────────────────────────
311
+
312
+ class TSRouteParser:
313
+ """Extract routes from Next.js, Express, NestJS via regex."""
314
+
315
+ NEXT_APP_ROUTE = re.compile(r'export\s+async\s+function\s+(GET|POST|PUT|PATCH|DELETE|HEAD)\s*\(')
316
+ NEXT_PAGES_ROUTE = re.compile(r'export\s+default\s+(?:async\s+)?function\s+handler')
317
+ EXPRESS_ROUTE = re.compile(r'(?:router|app)\.(get|post|put|patch|delete)\s*\(\s*[\'"]([^\'"]+)[\'"]')
318
+ NEST_DECORATOR = re.compile(r'@(Get|Post|Put|Patch|Delete)\s*\(\s*(?:[\'"]([^\'"]*)[\'"])?\s*\)')
319
+
320
+ def parse(self, files: list[Path]) -> list[dict]:
321
+ routes = []
322
+ for f in files:
323
+ if f.suffix not in ('.ts', '.tsx', '.js', '.jsx'):
324
+ continue
325
+ try:
326
+ source = f.read_text(encoding='utf-8', errors='replace')
327
+ except Exception:
328
+ continue
329
+ rel = str(f.relative_to(PROJECT_ROOT))
330
+
331
+ # Next.js App Router
332
+ for m in self.NEXT_APP_ROUTE.finditer(source):
333
+ line = source[:m.start()].count('\n') + 1
334
+ path = self._infer_next_path(f)
335
+ routes.append({'method': m.group(1), 'path': path, 'file': rel, 'line': line})
336
+
337
+ # Express
338
+ for m in self.EXPRESS_ROUTE.finditer(source):
339
+ line = source[:m.start()].count('\n') + 1
340
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
341
+
342
+ # NestJS
343
+ for m in self.NEST_DECORATOR.finditer(source):
344
+ line = source[:m.start()].count('\n') + 1
345
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2) or '/', 'file': rel, 'line': line})
346
+
347
+ return routes
348
+
349
+ def _infer_next_path(self, f: Path) -> str:
350
+ """Infer URL path from Next.js file location."""
351
+ parts = f.parts
352
+ try:
353
+ # app router: after 'app/'
354
+ app_idx = parts.index('app') if 'app' in parts else None
355
+ if app_idx:
356
+ seg = parts[app_idx+1:]
357
+ seg = [s for s in seg if not s.startswith('(') and s not in ('route.ts','route.js','page.tsx','page.ts')]
358
+ return '/' + '/'.join(seg) if seg else '/'
359
+ # pages router
360
+ pages_idx = parts.index('pages') if 'pages' in parts else None
361
+ if pages_idx:
362
+ seg = list(parts[pages_idx+1:])
363
+ seg[-1] = re.sub(r'\.(ts|tsx|js|jsx)$', '', seg[-1])
364
+ if seg[-1] in ('index',):
365
+ seg = seg[:-1]
366
+ return '/' + '/'.join(seg) if seg else '/'
367
+ except (ValueError, IndexError):
368
+ pass
369
+ return f'/{f.stem}'
370
+
371
+
372
+ class TSModelParser:
373
+ """Extract Prisma schema entities and TypeORM entities via regex."""
374
+
375
+ PRISMA_MODEL = re.compile(r'^model\s+(\w+)\s*\{([^}]+)\}', re.MULTILINE)
376
+ PRISMA_FIELD = re.compile(r'^\s+(\w+)\s+(\w+)', re.MULTILINE)
377
+ TYPEORM_ENTITY = re.compile(r'@Entity\s*\(')
378
+ CLASS_NAME = re.compile(r'class\s+(\w+)')
379
+
380
+ def parse(self, files: list[Path]) -> list[dict]:
381
+ models = []
382
+ for f in files:
383
+ rel = str(f.relative_to(PROJECT_ROOT))
384
+ # Prisma
385
+ if f.suffix == '.prisma':
386
+ try:
387
+ source = f.read_text(encoding='utf-8', errors='replace')
388
+ for m in self.PRISMA_MODEL.finditer(source):
389
+ fields = [
390
+ {'name': fm.group(1), 'type': fm.group(2)}
391
+ for fm in self.PRISMA_FIELD.finditer(m.group(2))
392
+ ]
393
+ models.append({'name': m.group(1), 'file': rel, 'columns': fields})
394
+ except Exception:
395
+ pass
396
+ # TypeORM / NestJS entities
397
+ elif f.suffix in ('.ts', '.tsx'):
398
+ try:
399
+ source = f.read_text(encoding='utf-8', errors='replace')
400
+ if self.TYPEORM_ENTITY.search(source):
401
+ cm = self.CLASS_NAME.search(source)
402
+ if cm:
403
+ models.append({'name': cm.group(1), 'file': rel, 'columns': []})
404
+ except Exception:
405
+ pass
406
+ return models
407
+
408
+
409
+ # ── Go ───────────────────────────────────────────────────────────────────────
410
+
411
+ class GoRouteParser:
412
+ GIN = re.compile(r'(?:router|r|v\d+|api)\.(GET|POST|PUT|PATCH|DELETE)\s*\(\s*"([^"]+)"')
413
+ CHI = re.compile(r'r\.(Get|Post|Put|Patch|Delete)\s*\(\s*"([^"]+)"')
414
+ STD = re.compile(r'(?:mux|http)\.HandleFunc\s*\(\s*"([^"]+)"')
415
+
416
+ def parse(self, files: list[Path]) -> list[dict]:
417
+ routes = []
418
+ for f in files:
419
+ if f.suffix != '.go':
420
+ continue
421
+ try:
422
+ source = f.read_text(encoding='utf-8', errors='replace')
423
+ except Exception:
424
+ continue
425
+ rel = str(f.relative_to(PROJECT_ROOT))
426
+ for m in self.GIN.finditer(source):
427
+ line = source[:m.start()].count('\n') + 1
428
+ routes.append({'method': m.group(1), 'path': m.group(2), 'file': rel, 'line': line})
429
+ for m in self.CHI.finditer(source):
430
+ line = source[:m.start()].count('\n') + 1
431
+ routes.append({'method': m.group(1).upper(), 'path': m.group(2), 'file': rel, 'line': line})
432
+ for m in self.STD.finditer(source):
433
+ line = source[:m.start()].count('\n') + 1
434
+ routes.append({'method': 'ANY', 'path': m.group(1), 'file': rel, 'line': line})
435
+ return routes
436
+
437
+
438
+ class GoModelParser:
439
+ STRUCT = re.compile(r'type\s+(\w+)\s+struct\s*\{([^}]+)\}', re.DOTALL)
440
+ FIELD = re.compile(r'^\s+(\w+)\s+(\S+)', re.MULTILINE)
441
+
442
+ def parse(self, files: list[Path]) -> list[dict]:
443
+ models = []
444
+ for f in files:
445
+ if f.suffix != '.go':
446
+ continue
447
+ try:
448
+ source = f.read_text(encoding='utf-8', errors='replace')
449
+ except Exception:
450
+ continue
451
+ rel = str(f.relative_to(PROJECT_ROOT))
452
+ for m in self.STRUCT.finditer(source):
453
+ # Only include structs that look like data models (have db/json tags)
454
+ body = m.group(2)
455
+ if 'db:' in body or 'json:' in body or 'gorm:' in body:
456
+ fields = [
457
+ {'name': fm.group(1), 'type': fm.group(2)}
458
+ for fm in self.FIELD.finditer(body)
459
+ if not fm.group(1).startswith('//')
460
+ ]
461
+ models.append({'name': m.group(1), 'file': rel, 'columns': fields})
462
+ return models
463
+
464
+
465
+ # ── Docker Compose ───────────────────────────────────────────────────────────
466
+
467
+ class DockerComposeParser:
468
+ def parse(self) -> list[dict]:
469
+ services = []
470
+ candidates = [
471
+ 'docker-compose.yml', 'docker-compose.yaml',
472
+ 'docker-compose.dev.yml', 'docker-compose.development.yml',
473
+ 'docker-compose.prod.yml', 'docker-compose.production.yml',
474
+ ]
475
+ for name in candidates:
476
+ path = PROJECT_ROOT / name
477
+ if not path.exists():
478
+ continue
479
+ try:
480
+ data = self._load(path)
481
+ if data and isinstance(data.get('services'), dict):
482
+ for svc_name, svc in data['services'].items():
483
+ services.append({
484
+ 'name': svc_name,
485
+ 'image': svc.get('image', svc.get('build', '(build)')),
486
+ 'ports': svc.get('ports', []),
487
+ 'depends_on': svc.get('depends_on', []),
488
+ 'environment_keys': list(svc.get('environment', {}).keys())
489
+ if isinstance(svc.get('environment'), dict)
490
+ else [],
491
+ 'source_file': name,
492
+ })
493
+ except Exception:
494
+ pass
495
+ return services
496
+
497
+ def _load(self, path: Path) -> dict | None:
498
+ text = path.read_text(encoding='utf-8', errors='replace')
499
+ if HAS_YAML:
500
+ return yaml.safe_load(text)
501
+ # Regex fallback: extract service names at minimum
502
+ services: dict = {'services': {}}
503
+ in_services = False
504
+ indent = 0
505
+ for line in text.splitlines():
506
+ if re.match(r'^services\s*:', line):
507
+ in_services = True
508
+ continue
509
+ if in_services:
510
+ m = re.match(r'^(\s+)(\w[\w-]*)\s*:', line)
511
+ if m:
512
+ lvl = len(m.group(1))
513
+ if indent == 0:
514
+ indent = lvl
515
+ if lvl == indent:
516
+ services['services'][m.group(2)] = {}
517
+ return services
518
+
519
+
520
+ # ── Env Parser ───────────────────────────────────────────────────────────────
521
+
522
+ class EnvParser:
523
+ SECRET_KEYS = re.compile(
524
+ r'(?i)(password|secret|token|key|auth|credential|private|cert)'
525
+ )
526
+
527
+ def parse(self) -> list[dict]:
528
+ entries = []
529
+ candidates = ['.env', '.env.example', '.env.local', '.env.development', '.env.sample']
530
+ for name in candidates:
531
+ path = PROJECT_ROOT / name
532
+ if not path.exists():
533
+ continue
534
+ for line in path.read_text(encoding='utf-8', errors='replace').splitlines():
535
+ line = line.strip()
536
+ if not line or line.startswith('#'):
537
+ continue
538
+ if '=' not in line:
539
+ continue
540
+ key, _, raw_val = line.partition('=')
541
+ key = key.strip()
542
+ val = raw_val.strip()
543
+ is_secret = bool(self.SECRET_KEYS.search(key))
544
+ entries.append({
545
+ 'key': key,
546
+ 'value': '[SECRET]' if is_secret else val[:80],
547
+ 'is_secret': is_secret,
548
+ 'source': name,
549
+ })
550
+ return entries
551
+
552
+
553
+ # ── Migration Parser ─────────────────────────────────────────────────────────
554
+
555
+ class MigrationParser:
556
+ def parse(self) -> list[dict]:
557
+ migrations = []
558
+ # Alembic
559
+ alembic_dirs = list(PROJECT_ROOT.rglob('versions'))
560
+ for d in alembic_dirs:
561
+ if not d.is_dir():
562
+ continue
563
+ for f in sorted(d.glob('*.py')):
564
+ try:
565
+ source = f.read_text(encoding='utf-8', errors='replace')
566
+ rev = re.search(r'revision\s*=\s*[\'"]([^\'"]+)[\'"]', source)
567
+ down = re.search(r'down_revision\s*=\s*[\'"]?([^\'")\n]+)[\'"]?', source)
568
+ msg = re.search(r'Create Date.*\n.*"""(.+?)"""', source, re.DOTALL)
569
+ tables = re.findall(r'op\.(?:create_table|drop_table|add_column)\s*\(\s*[\'"]([^\'"]+)[\'"]', source)
570
+ migrations.append({
571
+ 'revision': rev.group(1) if rev else f.stem,
572
+ 'down_revision': down.group(1).strip() if down else None,
573
+ 'tables_affected': list(set(tables)),
574
+ 'file': str(f.relative_to(PROJECT_ROOT)),
575
+ 'type': 'alembic',
576
+ })
577
+ except Exception:
578
+ pass
579
+
580
+ # SQL files
581
+ for f in sorted(PROJECT_ROOT.rglob('*.sql')):
582
+ if any(p in IGNORE_DIRS for p in f.parts):
583
+ continue
584
+ try:
585
+ source = f.read_text(encoding='utf-8', errors='replace')[:2000]
586
+ tables = re.findall(r'(?:CREATE|ALTER|DROP)\s+TABLE\s+(?:IF\s+NOT\s+EXISTS\s+)?(\w+)', source, re.IGNORECASE)
587
+ if tables:
588
+ migrations.append({
589
+ 'revision': f.stem,
590
+ 'tables_affected': list(set(tables)),
591
+ 'file': str(f.relative_to(PROJECT_ROOT)),
592
+ 'type': 'sql',
593
+ })
594
+ except Exception:
595
+ pass
596
+
597
+ return migrations
598
+
599
+
600
+ # ── Frontend Feature Scanner ─────────────────────────────────────────────────
601
+
602
+ class FrontendScanner:
603
+ FEATURE_DIRS = {'features', 'pages', 'views', 'screens', 'modules', 'app'}
604
+ COMPONENT_EXTS = {'.tsx', '.jsx'}
605
+ HOOK_PATTERN = re.compile(r'^use[A-Z]')
606
+ # Candidate bases to search for the FEATURE_DIRS, relative to PROJECT_ROOT.
607
+ # Covers repo-root and src/ layouts plus monorepos where the frontend is a
608
+ # subpackage (e.g. frontend/src/features), not just at the repo root.
609
+ SEARCH_PREFIXES = ('', 'src', 'frontend', 'frontend/src',
610
+ 'web', 'web/src', 'client', 'client/src', 'ui', 'ui/src')
611
+
612
+ def scan(self) -> list[dict]:
613
+ features, seen = [], set()
614
+ for prefix in self.SEARCH_PREFIXES:
615
+ base = PROJECT_ROOT / prefix if prefix else PROJECT_ROOT
616
+ for feat_dir_name in self.FEATURE_DIRS:
617
+ feat_dir = base / feat_dir_name
618
+ if not feat_dir.is_dir():
619
+ continue
620
+ for entry in sorted(feat_dir.iterdir()):
621
+ if not (entry.is_dir() and not entry.name.startswith('.')):
622
+ continue
623
+ rp = entry.resolve()
624
+ if rp in seen:
625
+ continue # already counted via another prefix
626
+ stats = self._scan_feature_dir(entry)
627
+ # Only emit dirs that actually contain frontend components, so a
628
+ # backend package named e.g. 'app/' is not misread as a feature.
629
+ if stats['component_count'] + stats['hook_count'] == 0:
630
+ continue
631
+ seen.add(rp)
632
+ stats['name'] = entry.name
633
+ stats['path'] = str(entry.relative_to(PROJECT_ROOT))
634
+ features.append(stats)
635
+ return features
636
+
637
+ def _scan_feature_dir(self, d: Path) -> dict:
638
+ components, hooks, stores, api_files = [], [], [], []
639
+ for f in d.rglob('*'):
640
+ if f.suffix in self.COMPONENT_EXTS:
641
+ if self.HOOK_PATTERN.match(f.stem):
642
+ hooks.append(f.name)
643
+ else:
644
+ components.append(f.name)
645
+ elif 'store' in f.name.lower() or 'slice' in f.name.lower():
646
+ stores.append(f.name)
647
+ elif 'api' in f.name.lower() or 'service' in f.name.lower() or 'client' in f.name.lower():
648
+ api_files.append(f.name)
649
+ return {
650
+ 'components': components,
651
+ 'hooks': hooks,
652
+ 'stores': stores,
653
+ 'api_files': api_files,
654
+ 'component_count': len(components),
655
+ 'hook_count': len(hooks),
656
+ }
657
+
658
+
659
+ # ── Import Chain Tracer ───────────────────────────────────────────────────────
660
+
661
+ class ImportChainTracer:
662
+ MAX_DEPTH = 5
663
+
664
+ def trace(self, routes: list[dict]) -> list[dict]:
665
+ chains = []
666
+ for route in routes[:30]: # limit to first 30 routes
667
+ f = PROJECT_ROOT / route.get('file', '')
668
+ if not f.exists() or f.suffix != '.py':
669
+ continue
670
+ chain = self._trace_file(f, depth=0)
671
+ if len(chain) > 1:
672
+ chains.append({
673
+ 'route': f"{route.get('method','?')} {route.get('path','?')}",
674
+ 'chain': ' → '.join(chain),
675
+ 'file': route.get('file'),
676
+ })
677
+ return chains
678
+
679
+ def _trace_file(self, f: Path, depth: int) -> list[str]:
680
+ if depth > self.MAX_DEPTH or not f.exists():
681
+ return []
682
+ result = [str(f.relative_to(PROJECT_ROOT))]
683
+ try:
684
+ source = f.read_text(encoding='utf-8', errors='replace')
685
+ tree = ast.parse(source)
686
+ for node in ast.walk(tree):
687
+ if isinstance(node, (ast.Import, ast.ImportFrom)):
688
+ if isinstance(node, ast.ImportFrom) and node.module:
689
+ # Attempt to resolve local import
690
+ parts = node.module.split('.')
691
+ candidate = PROJECT_ROOT / Path(*parts).with_suffix('.py')
692
+ if candidate.exists() and depth < self.MAX_DEPTH:
693
+ sub = self._trace_file(candidate, depth + 1)
694
+ if sub:
695
+ result.extend(sub[1:])
696
+ break
697
+ except Exception:
698
+ pass
699
+ return result
700
+
701
+
702
+ # ── Vocabulary Builder ────────────────────────────────────────────────────────
703
+
704
+ class VocabularyBuilder:
705
+ SKIP_WORDS = {'index', 'main', 'base', 'common', 'utils', 'helpers', 'types',
706
+ 'constants', 'config', 'lib', 'api', 'app', 'src', 'test', 'tests'}
707
+
708
+ def build(
709
+ self,
710
+ routes: list[dict],
711
+ models: list[dict],
712
+ schemas: list[dict],
713
+ features: list[dict],
714
+ stack: dict,
715
+ ) -> list[dict]:
716
+ vocab: dict[str, dict] = {}
717
+
718
+ # From features
719
+ for feat in features:
720
+ name = feat['name']
721
+ if name.lower() in self.SKIP_WORDS:
722
+ continue
723
+ aliases = self._name_to_aliases(name)
724
+ for alias in aliases:
725
+ self._add(vocab, alias, 'feature', feat.get('path', ''), f"{feat['component_count']} components")
726
+
727
+ # From models
728
+ for model in models:
729
+ aliases = self._name_to_aliases(model['name'])
730
+ for alias in aliases:
731
+ self._add(vocab, alias, 'model', model.get('file', ''), f"table/model: {model['name']}")
732
+
733
+ # From routes (group by path prefix)
734
+ route_groups: dict[str, list] = {}
735
+ for route in routes:
736
+ prefix = route.get('path', '/').split('/')[1] if '/' in route.get('path', '/') else route.get('path', '')
737
+ if prefix and prefix not in self.SKIP_WORDS:
738
+ route_groups.setdefault(prefix, []).append(route)
739
+ for prefix, group in route_groups.items():
740
+ aliases = self._name_to_aliases(prefix)
741
+ file_ex = group[0].get('file', '')
742
+ for alias in aliases:
743
+ self._add(vocab, alias, 'api', file_ex, f"{len(group)} routes")
744
+
745
+ # Merge learned vocabulary
746
+ learned = self._load_learned()
747
+ for alias, data in learned.items():
748
+ if alias not in vocab and data.get('score', 0) >= 5:
749
+ vocab[alias] = {
750
+ 'alias': alias,
751
+ 'type': 'learned',
752
+ 'location': ', '.join(data.get('targets', [])),
753
+ 'notes': f"learned from session (score: {data['score']:.1f})",
754
+ }
755
+
756
+ return list(vocab.values())
757
+
758
+ def _name_to_aliases(self, name: str) -> list[str]:
759
+ """Generate human aliases from a code name."""
760
+ # camelCase / PascalCase → words
761
+ words = re.sub(r'([A-Z])', r' \1', name).lower().split()
762
+ words = [w for w in words if w and w not in self.SKIP_WORDS]
763
+ if not words:
764
+ return []
765
+ phrase = ' '.join(words)
766
+ aliases = [phrase, name.lower()]
767
+ # e.g. 'deal-pipeline' → 'deal pipeline'
768
+ if '-' in name or '_' in name:
769
+ clean = re.sub(r'[-_]', ' ', name).lower()
770
+ aliases.append(clean)
771
+ return list(dict.fromkeys(aliases)) # dedupe, preserve order
772
+
773
+ def _add(self, vocab: dict, alias: str, type_: str, location: str, notes: str) -> None:
774
+ if alias and alias not in vocab:
775
+ vocab[alias] = {'alias': alias, 'type': type_, 'location': location, 'notes': notes}
776
+
777
+ def _load_learned(self) -> dict:
778
+ if LEARNED_VOC.exists():
779
+ try:
780
+ return json.loads(LEARNED_VOC.read_text())
781
+ except Exception:
782
+ pass
783
+ return {}
784
+
785
+
786
+ # ── Tools & Commands Scanner ──────────────────────────────────────────────────
787
+
788
+ class ToolsScanner:
789
+ def scan(self) -> list[dict]:
790
+ tools = []
791
+
792
+ # npm scripts
793
+ pkg = PROJECT_ROOT / 'package.json'
794
+ if pkg.exists():
795
+ try:
796
+ data = json.loads(pkg.read_text())
797
+ for name, cmd in (data.get('scripts') or {}).items():
798
+ tools.append({'name': name, 'command': f'npm run {name}', 'description': cmd, 'source': 'package.json'})
799
+ except Exception:
800
+ pass
801
+
802
+ # Makefile targets
803
+ makefile = PROJECT_ROOT / 'Makefile'
804
+ if makefile.exists():
805
+ for line in makefile.read_text(encoding='utf-8', errors='replace').splitlines():
806
+ m = re.match(r'^([a-zA-Z][a-zA-Z0-9_-]+)\s*:', line)
807
+ if m and not m.group(1).startswith('.'):
808
+ tools.append({'name': m.group(1), 'command': f'make {m.group(1)}', 'description': '', 'source': 'Makefile'})
809
+
810
+ # Poetry / pip scripts
811
+ for cfg in ['pyproject.toml', 'setup.cfg']:
812
+ path = PROJECT_ROOT / cfg
813
+ if path.exists():
814
+ text = path.read_text(encoding='utf-8', errors='replace')
815
+ for m in re.finditer(r'^\[tool\.poetry\.scripts\]\s*\n((?:\w.*\n)*)', text, re.MULTILINE):
816
+ for line in m.group(1).splitlines():
817
+ if '=' in line:
818
+ name = line.split('=')[0].strip()
819
+ tools.append({'name': name, 'command': name, 'description': '', 'source': cfg})
820
+
821
+ # Shell scripts at root
822
+ for f in PROJECT_ROOT.glob('*.sh'):
823
+ tools.append({'name': f.name, 'command': f'bash {f.name}', 'description': '', 'source': 'shell'})
824
+
825
+ # .claude/skills
826
+ skills_dir = PROJECT_ROOT / '.claude' / 'skills'
827
+ if skills_dir.is_dir():
828
+ for skill_dir in skills_dir.iterdir():
829
+ if (skill_dir / 'SKILL.md').exists():
830
+ tools.append({'name': f'/{skill_dir.name}', 'command': f'/{skill_dir.name}', 'description': 'Claude skill', 'source': 'skills'})
831
+
832
+ return tools
833
+
834
+
835
+ # ── Auth Config Scanner ───────────────────────────────────────────────────────
836
+
837
+ class AuthScanner:
838
+ def scan(self) -> dict:
839
+ info: dict[str, Any] = {'provider': 'unknown', 'files': [], 'patterns': []}
840
+
841
+ patterns_map = {
842
+ 'supabase': ['supabase', 'createClient', 'auth.signIn'],
843
+ 'next-auth': ['NextAuth', 'getSession', 'useSession', 'SessionProvider'],
844
+ 'clerk': ['ClerkProvider', 'useUser', '@clerk'],
845
+ 'auth0': ['Auth0Provider', 'useAuth0', '@auth0'],
846
+ 'jwt': ['jwt.sign', 'jwt.verify', 'create_access_token', 'decode_token'],
847
+ 'passport': ['passport.use', 'passport.authenticate'],
848
+ 'firebase': ['initializeApp', 'getAuth', 'signInWithEmailAndPassword'],
849
+ }
850
+
851
+ for f in PROJECT_ROOT.rglob('*'):
852
+ if any(p in IGNORE_DIRS for p in f.parts):
853
+ continue
854
+ if f.suffix not in ('.py', '.ts', '.tsx', '.js', '.jsx'):
855
+ continue
856
+ try:
857
+ source = f.read_text(encoding='utf-8', errors='replace')
858
+ except Exception:
859
+ continue
860
+ for provider, patterns in patterns_map.items():
861
+ if any(p in source for p in patterns):
862
+ info['provider'] = provider
863
+ rel = str(f.relative_to(PROJECT_ROOT))
864
+ if rel not in info['files']:
865
+ info['files'].append(rel)
866
+ info['patterns'].extend([p for p in patterns if p in source and p not in info['patterns']])
867
+
868
+ return info
869
+
870
+
871
+ # ── Reverse Proxy Scanner ─────────────────────────────────────────────────────
872
+
873
+ class ReverseProxyScanner:
874
+ def scan(self) -> list[dict]:
875
+ rules = []
876
+ # nginx
877
+ for f in PROJECT_ROOT.rglob('*.conf'):
878
+ if any(p in IGNORE_DIRS for p in f.parts):
879
+ continue
880
+ try:
881
+ source = f.read_text(encoding='utf-8', errors='replace')
882
+ for m in re.finditer(r'location\s+([^\s{]+)\s*\{[^}]*proxy_pass\s+([^;]+);', source, re.DOTALL):
883
+ rules.append({'type': 'nginx', 'location': m.group(1).strip(), 'upstream': m.group(2).strip(), 'file': str(f.relative_to(PROJECT_ROOT))})
884
+ except Exception:
885
+ pass
886
+ # Caddyfile
887
+ caddyfile = PROJECT_ROOT / 'Caddyfile'
888
+ if caddyfile.exists():
889
+ try:
890
+ source = caddyfile.read_text(encoding='utf-8', errors='replace')
891
+ for m in re.finditer(r'reverse_proxy\s+([^\n]+)', source):
892
+ rules.append({'type': 'caddy', 'location': '*', 'upstream': m.group(1).strip(), 'file': 'Caddyfile'})
893
+ except Exception:
894
+ pass
895
+ return rules
896
+
897
+
898
+ # ── Dead Code Detector ────────────────────────────────────────────────────────
899
+
900
+ class DeadCodeDetector:
901
+ def detect(self, vocab: list[dict]) -> list[dict]:
902
+ candidates = []
903
+ cutoff_date = '--since=6 months ago'
904
+
905
+ low_score_files = set()
906
+ if LEARNED_VOC.exists():
907
+ try:
908
+ learned = json.loads(LEARNED_VOC.read_text())
909
+ for alias, data in learned.items():
910
+ if data.get('score', 999) < 2:
911
+ low_score_files.update(data.get('targets', []))
912
+ except Exception:
913
+ pass
914
+
915
+ for filepath_str in low_score_files:
916
+ path = PROJECT_ROOT / filepath_str
917
+ if not path.exists():
918
+ continue
919
+ try:
920
+ result = subprocess.run(
921
+ ['git', '-C', str(PROJECT_ROOT), 'log', cutoff_date, '--', filepath_str],
922
+ capture_output=True, text=True, timeout=5
923
+ )
924
+ if not result.stdout.strip():
925
+ candidates.append({
926
+ 'file': filepath_str,
927
+ 'reason': 'No commits in 6 months + low vocabulary score',
928
+ })
929
+ except Exception:
930
+ pass
931
+
932
+ return candidates
933
+
934
+
935
+ # ╔══════════════════════════════════════════════════════════════════════════╗
936
+ # ║ SECTION WRITERS ║
937
+ # ╚══════════════════════════════════════════════════════════════════════════╝
938
+
939
+ def write_section(num: str, name: str, content: str) -> Path:
940
+ filename = f"{num}-{name}.md"
941
+ path = SECTIONS_DIR / filename
942
+ path.write_text(content, encoding='utf-8')
943
+ return path
944
+
945
+ def section_size_kb(path: Path) -> float:
946
+ return path.stat().st_size / 1024 if path.exists() else 0.0
947
+
948
+
949
+ def build_vocabulary_section(vocab: list[dict]) -> str:
950
+ lines = [
951
+ "# Section 01 — Vocabulary Translation Layer\n",
952
+ "> Maps human language to exact code locations. Auto-generated.\n\n",
953
+ "| Alias | Type | Location | Notes |",
954
+ "|-------|------|----------|-------|",
955
+ ]
956
+ for v in sorted(vocab, key=lambda x: x.get('alias', '')):
957
+ alias = v.get('alias', '').replace('|', '\\|')
958
+ type_ = v.get('type', '').replace('|', '\\|')
959
+ location = v.get('location', '').replace('|', '\\|')
960
+ notes = v.get('notes', '').replace('|', '\\|')
961
+ lines.append(f"| {alias} | {type_} | {location} | {notes} |")
962
+ if not vocab:
963
+ lines.append("| _(no vocabulary generated yet — add source code to populate)_ | | | |")
964
+ return '\n'.join(lines) + '\n'
965
+
966
+
967
+ def build_topology_section(services: list[dict]) -> str:
968
+ lines = [
969
+ "# Section 02 — Service Topology\n",
970
+ "> Docker services, ports, and dependencies.\n\n",
971
+ ]
972
+ if not services:
973
+ lines.append("_No docker-compose services detected._\n")
974
+ return '\n'.join(lines)
975
+ lines += ["| Service | Image | Ports | Depends On |",
976
+ "|---------|-------|-------|------------|"]
977
+ for svc in services:
978
+ ports = ', '.join(str(p) for p in svc.get('ports', []))
979
+ deps = ', '.join(svc.get('depends_on', []) if isinstance(svc.get('depends_on'), list)
980
+ else list(svc.get('depends_on', {}).keys()))
981
+ img = str(svc.get('image', '?'))[:50]
982
+ lines.append(f"| {svc['name']} | {img} | {ports} | {deps} |")
983
+ return '\n'.join(lines) + '\n'
984
+
985
+
986
+ def build_environment_section(env_entries: list[dict]) -> str:
987
+ lines = [
988
+ "# Section 03 — Environment Variables\n",
989
+ "> Key names and non-sensitive values only. Secrets are redacted.\n\n",
990
+ ]
991
+ if not env_entries:
992
+ lines.append("_No .env files found._\n")
993
+ return '\n'.join(lines)
994
+ lines += ["| Key | Value | Source |",
995
+ "|-----|-------|--------|"]
996
+ for e in env_entries:
997
+ val = '[SECRET]' if e.get('is_secret') else str(e.get('value', ''))[:60]
998
+ lines.append(f"| `{e['key']}` | `{val}` | {e.get('source', '')} |")
999
+ return '\n'.join(lines) + '\n'
1000
+
1001
+
1002
+ def build_routes_section(routes: list[dict]) -> str:
1003
+ lines = [
1004
+ "# Section 04 — API Routes\n\n",
1005
+ "| Method | Path | File | Line |",
1006
+ "|--------|------|------|------|",
1007
+ ]
1008
+ for r in sorted(routes, key=lambda x: (x.get('path',''), x.get('method',''))):
1009
+ lines.append(f"| `{r.get('method','?')}` | `{r.get('path','?')}` | {r.get('file','')} | {r.get('line','')} |")
1010
+ if not routes:
1011
+ lines.append("| _(no routes detected yet)_ | | | |")
1012
+ return '\n'.join(lines) + '\n'
1013
+
1014
+
1015
+ def build_models_section(models: list[dict]) -> str:
1016
+ lines = ["# Section 05 — Data Models\n\n"]
1017
+ if not models:
1018
+ lines.append("_No data models detected yet._\n")
1019
+ return '\n'.join(lines)
1020
+ for model in models:
1021
+ lines.append(f"## {model['name']}")
1022
+ lines.append(f"- **File**: `{model.get('file','?')}`")
1023
+ cols = model.get('columns', [])
1024
+ if cols:
1025
+ lines.append("- **Fields**:")
1026
+ for c in cols[:20]:
1027
+ lines.append(f" - `{c.get('name','?')}` ({c.get('type','?')})")
1028
+ lines.append('')
1029
+ return '\n'.join(lines)
1030
+
1031
+
1032
+ def build_schemas_section(schemas: list[dict]) -> str:
1033
+ lines = ["# Section 06 — Schemas / DTOs\n\n"]
1034
+ if not schemas:
1035
+ lines.append("_No schemas detected yet._\n")
1036
+ return '\n'.join(lines)
1037
+ for s in schemas:
1038
+ lines.append(f"## {s['name']}")
1039
+ lines.append(f"- **File**: `{s.get('file','?')}`")
1040
+ fields = s.get('fields', [])
1041
+ if fields:
1042
+ lines.append(f"- **Fields**: {', '.join(f'`{f}`' for f in fields[:15])}")
1043
+ lines.append('')
1044
+ return '\n'.join(lines)
1045
+
1046
+
1047
+ def build_services_section(routes: list[dict], models: list[dict]) -> str:
1048
+ lines = ["# Section 07 — Services\n\n",
1049
+ "_Service discovery is inferred from directory structure and imports._\n\n"]
1050
+ # Collect unique directories containing routes or models
1051
+ dirs: dict[str, int] = {}
1052
+ for item in routes + models:
1053
+ f = item.get('file', '')
1054
+ d = str(Path(f).parent) if f else ''
1055
+ if d and d != '.':
1056
+ dirs[d] = dirs.get(d, 0) + 1
1057
+ if dirs:
1058
+ lines += ["| Directory | Items |", "|-----------|-------|"]
1059
+ for d, count in sorted(dirs.items(), key=lambda x: -x[1]):
1060
+ lines.append(f"| `{d}` | {count} |")
1061
+ return '\n'.join(lines) + '\n'
1062
+
1063
+
1064
+ def build_background_jobs_section() -> str:
1065
+ lines = ["# Section 08 — Background Jobs\n\n"]
1066
+ jobs = []
1067
+ # Celery
1068
+ for f in PROJECT_ROOT.rglob('*.py'):
1069
+ if any(p in IGNORE_DIRS for p in f.parts):
1070
+ continue
1071
+ try:
1072
+ source = f.read_text(encoding='utf-8', errors='replace')
1073
+ for m in re.finditer(r'@(?:app|celery)\.task|@shared_task', source):
1074
+ # Find function after decorator
1075
+ fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
1076
+ if fn_m:
1077
+ line = source[:m.start()].count('\n') + 1
1078
+ jobs.append({'name': fn_m.group(1), 'type': 'celery', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
1079
+ except Exception:
1080
+ pass
1081
+ # Cron / APScheduler
1082
+ for f in PROJECT_ROOT.rglob('*.py'):
1083
+ if any(p in IGNORE_DIRS for p in f.parts):
1084
+ continue
1085
+ try:
1086
+ source = f.read_text(encoding='utf-8', errors='replace')
1087
+ for m in re.finditer(r'@scheduler\.scheduled_job|scheduler\.add_job|@cron', source):
1088
+ fn_m = re.search(r'def\s+(\w+)\s*\(', source[m.start():m.start()+200])
1089
+ if fn_m:
1090
+ line = source[:m.start()].count('\n') + 1
1091
+ jobs.append({'name': fn_m.group(1), 'type': 'scheduler', 'file': str(f.relative_to(PROJECT_ROOT)), 'line': line})
1092
+ except Exception:
1093
+ pass
1094
+
1095
+ if not jobs:
1096
+ lines.append("_No background jobs detected._\n")
1097
+ else:
1098
+ lines += ["| Job | Type | File | Line |", "|-----|------|------|------|"]
1099
+ for j in jobs:
1100
+ lines.append(f"| `{j['name']}` | {j['type']} | {j['file']} | {j.get('line','')} |")
1101
+ return '\n'.join(lines) + '\n'
1102
+
1103
+
1104
+ def build_frontend_section(features: list[dict]) -> str:
1105
+ lines = ["# Section 09 — Frontend Features\n\n"]
1106
+ if not features:
1107
+ lines.append("_No frontend feature directories detected._\n")
1108
+ return '\n'.join(lines)
1109
+ lines += ["| Feature | Components | Hooks | Stores | API Files |",
1110
+ "|---------|-----------|-------|--------|-----------|"]
1111
+ for f in features:
1112
+ lines.append(
1113
+ f"| `{f['path']}` | {f['component_count']} | {f['hook_count']} "
1114
+ f"| {len(f.get('stores',[]))} | {len(f.get('api_files',[]))} |"
1115
+ )
1116
+ return '\n'.join(lines) + '\n'
1117
+
1118
+
1119
+ def build_tools_section(tools: list[dict]) -> str:
1120
+ lines = ["# Section 10 — Tools & Commands\n\n",
1121
+ "| Name | Command | Source |",
1122
+ "|------|---------|--------|"]
1123
+ for t in tools:
1124
+ lines.append(f"| `{t['name']}` | `{t['command']}` | {t.get('source','')} |")
1125
+ if not tools:
1126
+ lines.append("| _(no tools detected)_ | | |")
1127
+ return '\n'.join(lines) + '\n'
1128
+
1129
+
1130
+ def build_migrations_section(migrations: list[dict]) -> str:
1131
+ lines = ["# Section 11 — Migrations\n\n"]
1132
+ if not migrations:
1133
+ lines.append("_No migrations detected._\n")
1134
+ return '\n'.join(lines)
1135
+ lines += ["| Revision | Tables Affected | Type | File |",
1136
+ "|----------|----------------|------|------|"]
1137
+ for m in migrations[-30:]: # last 30
1138
+ tables = ', '.join(m.get('tables_affected', []))[:60]
1139
+ lines.append(f"| `{m.get('revision','?')}` | {tables} | {m.get('type','?')} | {m.get('file','')} |")
1140
+ return '\n'.join(lines) + '\n'
1141
+
1142
+
1143
+ def build_import_chains_section(chains: list[dict]) -> str:
1144
+ lines = ["# Section 12 — Import Chains\n\n",
1145
+ "_Traces route → service → model → table for key endpoints._\n\n"]
1146
+ if not chains:
1147
+ lines.append("_No import chains traced (requires Python source files with routes)._\n")
1148
+ return '\n'.join(lines)
1149
+ for c in chains:
1150
+ lines.append(f"**{c['route']}**")
1151
+ lines.append(f"```\n{c['chain']}\n```\n")
1152
+ return '\n'.join(lines)
1153
+
1154
+
1155
+ def build_frontend_backend_section(routes: list[dict], features: list[dict]) -> str:
1156
+ lines = ["# Section 13 — Frontend → Backend Map\n\n",
1157
+ "_Maps frontend API service calls to backend route paths._\n\n"]
1158
+ mappings = []
1159
+ api_calls: list[tuple[str,str]] = []
1160
+
1161
+ for feat in features:
1162
+ for api_file in feat.get('api_files', []):
1163
+ full = PROJECT_ROOT / feat['path'] / api_file
1164
+ if full.exists():
1165
+ try:
1166
+ source = full.read_text(encoding='utf-8', errors='replace')
1167
+ for m in re.finditer(r'[\'"`](/api/[^\'"` \n]+)', source):
1168
+ api_calls.append((m.group(1), str(full.relative_to(PROJECT_ROOT))))
1169
+ except Exception:
1170
+ pass
1171
+
1172
+ route_paths = {r.get('path',''): r for r in routes}
1173
+ for call, fe_file in api_calls:
1174
+ if call in route_paths:
1175
+ be = route_paths[call]
1176
+ mappings.append({'frontend': fe_file, 'url': call, 'backend': be.get('file','?')})
1177
+
1178
+ if not mappings:
1179
+ lines.append("_No frontend→backend mappings detected yet._\n")
1180
+ else:
1181
+ lines += ["| Frontend File | URL | Backend File |",
1182
+ "|--------------|-----|--------------|"]
1183
+ for m in mappings:
1184
+ lines.append(f"| {m['frontend']} | `{m['url']}` | {m['backend']} |")
1185
+ return '\n'.join(lines) + '\n'
1186
+
1187
+
1188
+ def build_proxy_section(proxy_rules: list[dict]) -> str:
1189
+ lines = ["# Section 14 — Reverse Proxy\n\n"]
1190
+ if not proxy_rules:
1191
+ lines.append("_No reverse proxy configuration detected._\n")
1192
+ return '\n'.join(lines)
1193
+ lines += ["| Type | Location | Upstream | File |",
1194
+ "|------|----------|----------|------|"]
1195
+ for r in proxy_rules:
1196
+ lines.append(f"| {r['type']} | `{r['location']}` | `{r['upstream']}` | {r['file']} |")
1197
+ return '\n'.join(lines) + '\n'
1198
+
1199
+
1200
+ def build_auth_section(auth_info: dict) -> str:
1201
+ lines = ["# Section 15 — Auth Configuration\n\n",
1202
+ f"**Provider**: {auth_info.get('provider','unknown')}\n\n"]
1203
+ files = auth_info.get('files', [])
1204
+ if files:
1205
+ lines.append("**Auth files:**")
1206
+ for f in files[:10]:
1207
+ lines.append(f"- `{f}`")
1208
+ patterns = auth_info.get('patterns', [])
1209
+ if patterns:
1210
+ lines.append(f"\n**Detected patterns**: {', '.join(f'`{p}`' for p in patterns[:10])}")
1211
+ return '\n'.join(lines) + '\n'
1212
+
1213
+
1214
+ def build_infra_section(stack: dict, services: list[dict]) -> str:
1215
+ lines = ["# Section 16 — Infrastructure Profile\n\n",
1216
+ f"| Property | Value |",
1217
+ "|----------|-------|",
1218
+ f"| **Language** | {stack.get('language','?')} |",
1219
+ f"| **Framework** | {stack.get('framework','?')} |",
1220
+ f"| **Database** | {stack.get('database','?')} |",
1221
+ f"| **ORM** | {stack.get('orm','?')} |",
1222
+ f"| **Auth** | {stack.get('auth','?')} |",
1223
+ f"| **Package Manager** | {stack.get('package_manager','?')} |",
1224
+ f"| **Infrastructure** | {stack.get('infrastructure','?')} |",
1225
+ f"| **Services** | {len(services)} docker services |",
1226
+ ""]
1227
+ return '\n'.join(lines)
1228
+
1229
+
1230
+ def build_learned_vocab_section() -> str:
1231
+ lines = ["# Section 17 — Learned Vocabulary\n\n",
1232
+ "_Aliases mined from Claude Code session history. Score = frequency × recency._\n\n"]
1233
+ learned = {}
1234
+ if LEARNED_VOC.exists():
1235
+ try:
1236
+ learned = json.loads(LEARNED_VOC.read_text())
1237
+ except Exception:
1238
+ pass
1239
+ if not learned:
1240
+ lines.append("_No session-mined vocabulary yet. Accumulates over time._\n")
1241
+ return '\n'.join(lines)
1242
+ lines += ["| Alias | Score | Targets | Last Seen |",
1243
+ "|-------|-------|---------|-----------|"]
1244
+ for alias, data in sorted(learned.items(), key=lambda x: -x[1].get('score', 0)):
1245
+ if data.get('score', 0) >= 2:
1246
+ targets = ', '.join(data.get('targets', []))[:60]
1247
+ lines.append(f"| {alias} | {data.get('score',0):.1f} | {targets} | {data.get('last_seen','?')} |")
1248
+ return '\n'.join(lines) + '\n'
1249
+
1250
+
1251
+ def build_dead_code_section(candidates: list[dict]) -> str:
1252
+ lines = ["# Section 18 — Dead Code Candidates\n\n",
1253
+ "> Files flagged for human review: no recent commits AND low vocabulary score.\n",
1254
+ "> **Do not auto-delete.** Review before removing.\n\n"]
1255
+ if not candidates:
1256
+ lines.append("_No dead code candidates detected._\n")
1257
+ return '\n'.join(lines)
1258
+ for c in candidates:
1259
+ lines.append(f"- `{c['file']}` — {c.get('reason','')}")
1260
+ return '\n'.join(lines) + '\n'
1261
+
1262
+
1263
+ def build_doc_pointers_section() -> str:
1264
+ lines = ["# Section 19 — Documentation Pointers\n\n"]
1265
+ docs = []
1266
+ doc_dirs = ['docs', 'doc', 'documentation', 'wiki', '.docs']
1267
+ doc_exts = {'.md', '.rst', '.txt', '.adoc'}
1268
+
1269
+ for doc_dir in doc_dirs:
1270
+ d = PROJECT_ROOT / doc_dir
1271
+ if d.is_dir():
1272
+ for f in sorted(d.rglob('*')):
1273
+ if f.is_file() and f.suffix in doc_exts:
1274
+ docs.append(str(f.relative_to(PROJECT_ROOT)))
1275
+
1276
+ # Root-level docs
1277
+ for f in PROJECT_ROOT.glob('*.md'):
1278
+ docs.append(str(f.relative_to(PROJECT_ROOT)))
1279
+
1280
+ if not docs:
1281
+ lines.append("_No documentation files found._\n")
1282
+ else:
1283
+ for d in docs[:30]:
1284
+ lines.append(f"- [`{d}`]({d})")
1285
+ return '\n'.join(lines) + '\n'
1286
+
1287
+
1288
+ # ╔══════════════════════════════════════════════════════════════════════════╗
1289
+ # ║ PROJECT MAP TOC ║
1290
+ # ╚══════════════════════════════════════════════════════════════════════════╝
1291
+
1292
+ def build_project_map(
1293
+ stack: dict,
1294
+ routes: list[dict],
1295
+ models: list[dict],
1296
+ schemas: list[dict],
1297
+ features: list[dict],
1298
+ migrations: list[dict],
1299
+ services: list[dict],
1300
+ vocab: list[dict],
1301
+ section_files: list[tuple[str, Path]],
1302
+ ) -> str:
1303
+ now = datetime.now().strftime('%Y-%m-%d %H:%M')
1304
+ name = stack.get('name', PROJECT_ROOT.name)
1305
+ slug = stack.get('slug', name)
1306
+
1307
+ lines = [
1308
+ f"# {name} — Project Map\n",
1309
+ f"> Auto-generated by `generate.py` on {now}. Do not edit manually.\n\n",
1310
+ f"## Stats\n",
1311
+ f"| Metric | Count |",
1312
+ f"|--------|-------|",
1313
+ f"| API Routes | {len(routes)} |",
1314
+ f"| Data Models | {len(models)} |",
1315
+ f"| Schemas/DTOs | {len(schemas)} |",
1316
+ f"| Frontend Features | {len(features)} |",
1317
+ f"| Migrations | {len(migrations)} |",
1318
+ f"| Docker Services | {len(services)} |",
1319
+ f"| Vocabulary Entries | {len(vocab)} |",
1320
+ f"| Stack | {stack.get('language','?')} / {stack.get('framework','?')} |",
1321
+ "",
1322
+ "## Section Index\n",
1323
+ "| # | Section | Size | When to Read |",
1324
+ "|---|---------|------|--------------|",
1325
+ ]
1326
+
1327
+ WHEN_TO_READ = {
1328
+ '01': 'Any task — start here if you don\'t know where the code lives',
1329
+ '02': 'Debugging connectivity, adding a service, understanding ports',
1330
+ '03': 'Environment setup, missing vars, config issues',
1331
+ '04': 'Adding/editing API endpoints, checking what routes exist',
1332
+ '05': 'Changing database schema, adding fields, understanding relations',
1333
+ '06': 'Adding DTOs, changing request/response shapes',
1334
+ '07': 'Adding service logic, understanding service boundaries',
1335
+ '08': 'Working with background jobs, queues, scheduled tasks',
1336
+ '09': 'Frontend feature work, understanding UI structure',
1337
+ '10': 'Available commands, scripts, developer tooling',
1338
+ '11': 'Database migrations, schema history',
1339
+ '12': 'Tracing data flow from HTTP request to DB',
1340
+ '13': 'Understanding which frontend calls which backend endpoint',
1341
+ '14': 'Proxy routing, nginx/caddy config',
1342
+ '15': 'Auth flow, sessions, permissions',
1343
+ '16': 'Infrastructure overview, tech stack summary',
1344
+ '17': 'Vocabulary learned from past sessions',
1345
+ '18': 'Dead code review',
1346
+ '19': 'Finding documentation, READMEs, wikis',
1347
+ }
1348
+
1349
+ for name_part, path in section_files:
1350
+ num = name_part.split('-')[0]
1351
+ display = name_part.replace('-', ' ').title()
1352
+ size_kb = section_size_kb(path)
1353
+ when = WHEN_TO_READ.get(num, '')
1354
+ lines.append(f"| [{num}](sections/{path.name}) | {display} | {size_kb:.1f} KB | {when} |")
1355
+
1356
+ lines += [
1357
+ "",
1358
+ "## Quick Routing\n",
1359
+ "| Task | Read Sections |",
1360
+ "|------|--------------|",
1361
+ "| Feature / UX work | 01 → 09 → 04 |",
1362
+ "| Add model or field | 05 → 06 → 12 |",
1363
+ "| Troubleshoot error | 02 → 03 → 14 |",
1364
+ "| Infrastructure / scaling | 16 → 02 |",
1365
+ "| Auth / security | 15 → 19 |",
1366
+ "| What tools exist | 10 |",
1367
+ "| Background jobs | 08 |",
1368
+ "| Migration history | 11 |",
1369
+ "",
1370
+ f"## Regenerate\n",
1371
+ "```bash",
1372
+ "python .claude/project-map/generate.py # skip if unchanged",
1373
+ "python .claude/project-map/generate.py --force # always regenerate",
1374
+ "```",
1375
+ ]
1376
+ return '\n'.join(lines) + '\n'
1377
+
1378
+
1379
+ # ╔══════════════════════════════════════════════════════════════════════════╗
1380
+ # ║ MAIN ║
1381
+ # ╚══════════════════════════════════════════════════════════════════════════╝
1382
+
1383
+ def main() -> None:
1384
+ parser = argparse.ArgumentParser(description='Babel Fish — generate project map')
1385
+ parser.add_argument('--force', action='store_true', help='Force regeneration even if checksums match')
1386
+ parser.add_argument('--project-root', type=Path, default=None, help='Override project root')
1387
+ parser.add_argument('--stack-json', type=Path, default=None, help='Path to stack.json from detect-stack.sh')
1388
+ args = parser.parse_args()
1389
+
1390
+ global PROJECT_ROOT
1391
+ if args.project_root:
1392
+ PROJECT_ROOT = args.project_root.resolve()
1393
+
1394
+ print(f"[generate] Project root: {PROJECT_ROOT}")
1395
+
1396
+ # 1. Checksum check
1397
+ watched = collect_watched_files()
1398
+ checksum = compute_checksum(watched)
1399
+
1400
+ if not args.force and is_unchanged(checksum):
1401
+ print("[generate] ✓ No changes detected — skipping regeneration (use --force to override)")
1402
+ sys.exit(0)
1403
+
1404
+ # 2. Load stack
1405
+ stack = load_stack(args.stack_json)
1406
+ print(f"[generate] Stack: {stack['language']} / {stack['framework']}")
1407
+
1408
+ # 3. Collect all relevant files
1409
+ py_files = [f for f in watched if f.suffix == '.py']
1410
+ ts_files = [f for f in watched if f.suffix in ('.ts','.tsx','.js','.jsx')]
1411
+ go_files = [f for f in watched if f.suffix == '.go']
1412
+
1413
+ # 4. Parse
1414
+ print("[generate] Parsing routes...")
1415
+ routes: list[dict] = []
1416
+ if stack['language'] in ('python', 'unknown'):
1417
+ routes.extend(PythonRouteParser().parse(py_files))
1418
+ if stack['language'] in ('typescript', 'javascript', 'unknown'):
1419
+ routes.extend(TSRouteParser().parse(ts_files))
1420
+ if stack['language'] in ('go', 'unknown'):
1421
+ routes.extend(GoRouteParser().parse(go_files))
1422
+
1423
+ print("[generate] Parsing models...")
1424
+ models: list[dict] = []
1425
+ if stack['language'] in ('python', 'unknown'):
1426
+ models.extend(PythonModelParser().parse(py_files))
1427
+ if stack['language'] in ('typescript', 'javascript', 'unknown'):
1428
+ models.extend(TSModelParser().parse(ts_files + [f for f in watched if f.suffix == '.prisma']))
1429
+ if stack['language'] in ('go', 'unknown'):
1430
+ models.extend(GoModelParser().parse(go_files))
1431
+
1432
+ print("[generate] Parsing schemas...")
1433
+ schemas = PythonSchemaParser().parse(py_files)
1434
+
1435
+ print("[generate] Scanning environment...")
1436
+ services = DockerComposeParser().parse()
1437
+ env_entries = EnvParser().parse()
1438
+ migrations = MigrationParser().parse()
1439
+ features = FrontendScanner().scan()
1440
+ tools = ToolsScanner().scan()
1441
+ auth_info = AuthScanner().scan()
1442
+ proxy_rules = ReverseProxyScanner().scan()
1443
+
1444
+ print("[generate] Building vocabulary...")
1445
+ vocab = VocabularyBuilder().build(routes, models, schemas, features, stack)
1446
+
1447
+ print("[generate] Tracing import chains...")
1448
+ chains = ImportChainTracer().trace(routes)
1449
+
1450
+ print("[generate] Detecting dead code candidates...")
1451
+ dead_code = DeadCodeDetector().detect(vocab)
1452
+
1453
+ # 5. Write sections
1454
+ print("[generate] Writing sections...")
1455
+ section_files: list[tuple[str, Path]] = [
1456
+ ('01-vocabulary', write_section('01', 'vocabulary', build_vocabulary_section(vocab))),
1457
+ ('02-service-topology', write_section('02', 'service-topology', build_topology_section(services))),
1458
+ ('03-environment', write_section('03', 'environment', build_environment_section(env_entries))),
1459
+ ('04-api-routes', write_section('04', 'api-routes', build_routes_section(routes))),
1460
+ ('05-data-models', write_section('05', 'data-models', build_models_section(models))),
1461
+ ('06-schemas', write_section('06', 'schemas', build_schemas_section(schemas))),
1462
+ ('07-services', write_section('07', 'services', build_services_section(routes, models))),
1463
+ ('08-background-jobs', write_section('08', 'background-jobs', build_background_jobs_section())),
1464
+ ('09-frontend-features', write_section('09', 'frontend-features', build_frontend_section(features))),
1465
+ ('10-tools-commands', write_section('10', 'tools-commands', build_tools_section(tools))),
1466
+ ('11-migrations', write_section('11', 'migrations', build_migrations_section(migrations))),
1467
+ ('12-import-chains', write_section('12', 'import-chains', build_import_chains_section(chains))),
1468
+ ('13-frontend-backend-map',write_section('13', 'frontend-backend-map',build_frontend_backend_section(routes, features))),
1469
+ ('14-reverse-proxy', write_section('14', 'reverse-proxy', build_proxy_section(proxy_rules))),
1470
+ ('15-auth-config', write_section('15', 'auth-config', build_auth_section(auth_info))),
1471
+ ('16-infra-profile', write_section('16', 'infra-profile', build_infra_section(stack, services))),
1472
+ ('17-learned-vocabulary', write_section('17', 'learned-vocabulary', build_learned_vocab_section())),
1473
+ ('18-dead-code', write_section('18', 'dead-code', build_dead_code_section(dead_code))),
1474
+ ('19-doc-pointers', write_section('19', 'doc-pointers', build_doc_pointers_section())),
1475
+ ]
1476
+
1477
+ # 6. Write PROJECT_MAP.md
1478
+ print("[generate] Writing PROJECT_MAP.md...")
1479
+ project_map = build_project_map(stack, routes, models, schemas, features, migrations, services, vocab, section_files)
1480
+ (MAP_DIR / 'PROJECT_MAP.md').write_text(project_map, encoding='utf-8')
1481
+
1482
+ # 7. Update checksums
1483
+ save_checksums({'input_hash': checksum, 'generated_at': datetime.now().isoformat(), 'route_count': len(routes), 'model_count': len(models)})
1484
+
1485
+ # 8. Ensure learned-vocabulary.json exists
1486
+ if not LEARNED_VOC.exists():
1487
+ LEARNED_VOC.write_text('{}', encoding='utf-8')
1488
+
1489
+ total_kb = sum(section_size_kb(p) for _, p in section_files)
1490
+ print(f"[generate] ✓ Done — {len(routes)} routes, {len(models)} models, {len(vocab)} vocab entries, {total_kb:.1f} KB total")
1491
+
1492
+
1493
+ if __name__ == '__main__':
1494
+ main()