@theglitchking/babel-fish 1.0.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/install.sh +666 -0
- package/.claude/project-map/PROJECT_MAP.md +61 -0
- package/.claude/project-map/checksums.json +6 -0
- package/.claude/project-map/generate.py +1481 -0
- package/.claude/project-map/grader.py +583 -0
- package/.claude/project-map/learned-vocabulary.json +1 -0
- package/.claude/project-map/mine-sessions.py +419 -0
- package/.claude/project-map/reports/install-report.md +54 -0
- package/.claude/project-map/reports/iteration-01-report.md +54 -0
- package/.claude/project-map/reports/iteration-01-score.json +7 -0
- package/.claude/project-map/sections/01-vocabulary.md +8 -0
- package/.claude/project-map/sections/02-service-topology.md +6 -0
- package/.claude/project-map/sections/03-environment.md +6 -0
- package/.claude/project-map/sections/04-api-routes.md +6 -0
- package/.claude/project-map/sections/05-data-models.md +4 -0
- package/.claude/project-map/sections/06-schemas.md +4 -0
- package/.claude/project-map/sections/07-services.md +6 -0
- package/.claude/project-map/sections/08-background-jobs.md +5 -0
- package/.claude/project-map/sections/09-frontend-features.md +4 -0
- package/.claude/project-map/sections/10-tools-commands.md +8 -0
- package/.claude/project-map/sections/11-migrations.md +4 -0
- package/.claude/project-map/sections/12-import-chains.md +7 -0
- package/.claude/project-map/sections/13-frontend-backend-map.md +8 -0
- package/.claude/project-map/sections/14-reverse-proxy.md +4 -0
- package/.claude/project-map/sections/15-auth-config.md +6 -0
- package/.claude/project-map/sections/16-infra-profile.md +13 -0
- package/.claude/project-map/sections/17-learned-vocabulary.md +7 -0
- package/.claude/project-map/sections/18-dead-code.md +9 -0
- package/.claude/project-map/sections/19-doc-pointers.md +5 -0
- package/.claude/project-map/stack.json +12 -0
- package/.claude/rules/operational-runbook.md +40 -0
- package/.claude/rules/project-vocabulary.md +25 -0
- package/.claude/scripts/detect-stack.sh +222 -0
- package/.claude/scripts/ensure-python.sh +100 -0
- package/.claude/scripts/statusline.sh +27 -0
- package/.claude/scripts/validate.sh +59 -0
- package/.claude/settings.json +6 -0
- package/.claude/skills/babel-fish-developer-skill/SKILL.md +56 -0
- package/.claude/templates/SKILL.md.template +56 -0
- package/.claude/templates/operational-runbook.md.template +40 -0
- package/.claude/templates/project-vocabulary.md.template +25 -0
- package/.claude-plugin/marketplace.json +2 -2
- package/.claude-plugin/plugin.json +1 -1
- package/.githooks/install.sh +4 -0
- package/.githooks/pre-commit +22 -0
- package/CHANGELOG.md +72 -0
- package/README.md +18 -0
- package/bin/babel-fish.js +78 -71
- package/commands/policy.md +16 -0
- package/commands/relink.md +6 -0
- package/commands/status.md +6 -0
- package/commands/update.md +6 -0
- package/hooks/hooks.json +15 -0
- package/hooks/session-start.js +11 -0
- package/package.json +17 -4
- package/scripts/link-skills.js +31 -0
|
@@ -0,0 +1,419 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
mine-sessions.py — Session Vocabulary Miner
|
|
4
|
+
Parses Claude Code conversation JSONL files from ~/.claude/projects/<slug>/
|
|
5
|
+
to extract user-language → file-path aliases.
|
|
6
|
+
|
|
7
|
+
Pattern: user said "the numbers page" → Claude opened DealAnalyzerV2.tsx
|
|
8
|
+
= learned alias: "numbers page" → DealAnalyzerV2.tsx
|
|
9
|
+
|
|
10
|
+
Scoring: frequency × recency_weight
|
|
11
|
+
- Sessions from last 30 days: weight 1.0
|
|
12
|
+
- 31-60 days: weight 0.5
|
|
13
|
+
- 61-90 days: weight 0.25
|
|
14
|
+
- >90 days: dropped
|
|
15
|
+
|
|
16
|
+
Minimum score of 5 to appear in output (filters one-off mentions).
|
|
17
|
+
|
|
18
|
+
Usage:
|
|
19
|
+
python mine-sessions.py [--project-root PATH] [--dry-run] [--verbose]
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import argparse
|
|
24
|
+
import json
|
|
25
|
+
import os
|
|
26
|
+
import re
|
|
27
|
+
import sys
|
|
28
|
+
from datetime import datetime, timedelta, timezone
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
# ── Paths ────────────────────────────────────────────────────────────────────
|
|
32
|
+
SCRIPT_DIR = Path(__file__).parent
|
|
33
|
+
LEARNED_VOC = SCRIPT_DIR / "learned-vocabulary.json"
|
|
34
|
+
PROJECT_ROOT = SCRIPT_DIR.parent.parent
|
|
35
|
+
|
|
36
|
+
# ── Config ───────────────────────────────────────────────────────────────────
|
|
37
|
+
MIN_SCORE = 5.0 # Minimum score to include in vocabulary
|
|
38
|
+
RECENCY_WINDOWS = [ # (days_threshold, weight)
|
|
39
|
+
(30, 1.00),
|
|
40
|
+
(60, 0.50),
|
|
41
|
+
(90, 0.25),
|
|
42
|
+
]
|
|
43
|
+
MAX_ALIAS_WORDS = 6 # Ignore user phrases longer than this
|
|
44
|
+
MIN_ALIAS_WORDS = 1
|
|
45
|
+
STOP_WORDS = {
|
|
46
|
+
'the', 'a', 'an', 'this', 'that', 'these', 'those', 'it', 'its',
|
|
47
|
+
'and', 'or', 'but', 'in', 'on', 'at', 'to', 'for', 'of', 'with',
|
|
48
|
+
'is', 'are', 'was', 'were', 'be', 'been', 'has', 'have', 'had',
|
|
49
|
+
'do', 'does', 'did', 'will', 'would', 'could', 'should', 'can',
|
|
50
|
+
'i', 'me', 'my', 'we', 'our', 'you', 'your', 'he', 'she', 'they',
|
|
51
|
+
'please', 'just', 'also', 'now', 'then', 'so', 'ok', 'yes', 'no',
|
|
52
|
+
'what', 'how', 'why', 'when', 'where', 'which', 'who',
|
|
53
|
+
'file', 'files', 'code', 'function', 'method', 'class', 'module',
|
|
54
|
+
'here', 'there', 'look', 'see', 'check', 'let', 'make', 'get',
|
|
55
|
+
'add', 'update', 'change', 'fix', 'show', 'find', 'help', 'need',
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
# Tool call names that indicate a file was opened/used
|
|
59
|
+
FILE_TOOL_NAMES = {
|
|
60
|
+
'Read', 'Edit', 'Write', 'Grep', 'Glob',
|
|
61
|
+
'read_file', 'edit_file', 'write_file',
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
# ── Recency weight ────────────────────────────────────────────────────────────
|
|
65
|
+
|
|
66
|
+
def recency_weight(session_date: datetime) -> float:
|
|
67
|
+
now = datetime.now(tz=timezone.utc)
|
|
68
|
+
if session_date.tzinfo is None:
|
|
69
|
+
session_date = session_date.replace(tzinfo=timezone.utc)
|
|
70
|
+
age_days = (now - session_date).days
|
|
71
|
+
for threshold, weight in RECENCY_WINDOWS:
|
|
72
|
+
if age_days <= threshold:
|
|
73
|
+
return weight
|
|
74
|
+
return 0.0 # Too old — drop
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ── Session discovery ─────────────────────────────────────────────────────────
|
|
78
|
+
|
|
79
|
+
def find_session_files(project_root: Path) -> list[Path]:
|
|
80
|
+
"""Find Claude Code JSONL session files for this project."""
|
|
81
|
+
candidates: list[Path] = []
|
|
82
|
+
|
|
83
|
+
# ~/.claude/projects/ uses a path-encoded slug
|
|
84
|
+
# e.g. /mnt/e/the-glitch-kingdom/babel-fish → -mnt-e-the-glitch-kingdom-babel-fish
|
|
85
|
+
encoded = str(project_root).replace('/', '-').lstrip('-')
|
|
86
|
+
claude_projects = Path.home() / '.claude' / 'projects'
|
|
87
|
+
|
|
88
|
+
if not claude_projects.exists():
|
|
89
|
+
return []
|
|
90
|
+
|
|
91
|
+
# Try exact match first
|
|
92
|
+
exact = claude_projects / encoded
|
|
93
|
+
if exact.is_dir():
|
|
94
|
+
candidates.extend(sorted(exact.glob('*.jsonl')))
|
|
95
|
+
|
|
96
|
+
# Also try fuzzy match on project name
|
|
97
|
+
project_name = project_root.name.lower()
|
|
98
|
+
for d in claude_projects.iterdir():
|
|
99
|
+
if d.is_dir() and project_name in d.name.lower() and d != exact:
|
|
100
|
+
candidates.extend(sorted(d.glob('*.jsonl')))
|
|
101
|
+
|
|
102
|
+
return candidates
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# ── JSONL parser ──────────────────────────────────────────────────────────────
|
|
106
|
+
|
|
107
|
+
def parse_session(path: Path) -> list[dict]:
|
|
108
|
+
"""Parse a JSONL session file into a list of message dicts."""
|
|
109
|
+
messages = []
|
|
110
|
+
try:
|
|
111
|
+
for line in path.read_text(encoding='utf-8', errors='replace').splitlines():
|
|
112
|
+
line = line.strip()
|
|
113
|
+
if not line:
|
|
114
|
+
continue
|
|
115
|
+
try:
|
|
116
|
+
obj = json.loads(line)
|
|
117
|
+
messages.append(obj)
|
|
118
|
+
except json.JSONDecodeError:
|
|
119
|
+
pass
|
|
120
|
+
except Exception:
|
|
121
|
+
pass
|
|
122
|
+
return messages
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def extract_session_date(messages: list[dict]) -> datetime | None:
|
|
126
|
+
"""Extract the timestamp of the first message in a session."""
|
|
127
|
+
for msg in messages:
|
|
128
|
+
ts = msg.get('timestamp') or msg.get('created_at') or msg.get('ts')
|
|
129
|
+
if ts:
|
|
130
|
+
try:
|
|
131
|
+
return datetime.fromisoformat(str(ts).replace('Z', '+00:00'))
|
|
132
|
+
except ValueError:
|
|
133
|
+
pass
|
|
134
|
+
return None
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
# ── Alias extraction ──────────────────────────────────────────────────────────
|
|
138
|
+
|
|
139
|
+
def extract_file_paths_from_tool_calls(messages: list[dict]) -> list[str]:
|
|
140
|
+
"""Extract file paths from tool use calls in a message sequence."""
|
|
141
|
+
paths = []
|
|
142
|
+
for msg in messages:
|
|
143
|
+
# Handle various JSONL formats
|
|
144
|
+
content = msg.get('content') or []
|
|
145
|
+
if isinstance(content, str):
|
|
146
|
+
continue
|
|
147
|
+
for block in content:
|
|
148
|
+
if not isinstance(block, dict):
|
|
149
|
+
continue
|
|
150
|
+
if block.get('type') not in ('tool_use', 'tool_result'):
|
|
151
|
+
continue
|
|
152
|
+
tool_name = block.get('name', '')
|
|
153
|
+
if tool_name not in FILE_TOOL_NAMES:
|
|
154
|
+
continue
|
|
155
|
+
# Extract file_path from input
|
|
156
|
+
inp = block.get('input') or {}
|
|
157
|
+
if isinstance(inp, dict):
|
|
158
|
+
fp = inp.get('file_path') or inp.get('path') or inp.get('pattern')
|
|
159
|
+
if fp and isinstance(fp, str):
|
|
160
|
+
# Normalize to relative path
|
|
161
|
+
try:
|
|
162
|
+
rel = Path(fp).relative_to(PROJECT_ROOT)
|
|
163
|
+
paths.append(str(rel))
|
|
164
|
+
except ValueError:
|
|
165
|
+
# Already relative or different root
|
|
166
|
+
if not fp.startswith('/'):
|
|
167
|
+
paths.append(fp)
|
|
168
|
+
return paths
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def extract_user_phrases(text: str) -> list[str]:
|
|
172
|
+
"""
|
|
173
|
+
Extract candidate alias phrases from user message text.
|
|
174
|
+
Returns noun phrases and quoted strings likely to be feature references.
|
|
175
|
+
"""
|
|
176
|
+
phrases = []
|
|
177
|
+
|
|
178
|
+
# Quoted strings are high-confidence aliases
|
|
179
|
+
for m in re.finditer(r'["\']([^"\']{3,40})["\']', text):
|
|
180
|
+
phrases.append(m.group(1).lower().strip())
|
|
181
|
+
|
|
182
|
+
# "the X" / "the X page/screen/section/tab/view/panel/modal/form/button"
|
|
183
|
+
for m in re.finditer(
|
|
184
|
+
r'\bthe\s+([\w\s-]{2,40?}?)\s*(?:page|screen|section|tab|view|panel|modal|form|button|component|widget|dashboard|list|table|chart|graph|map|sidebar|header|footer|nav|menu)\b',
|
|
185
|
+
text, re.IGNORECASE
|
|
186
|
+
):
|
|
187
|
+
phrases.append(m.group(1).lower().strip())
|
|
188
|
+
|
|
189
|
+
# "X feature" / "X functionality" / "X system" / "X module"
|
|
190
|
+
for m in re.finditer(
|
|
191
|
+
r'\b([\w\s-]{2,30?}?)\s+(?:feature|functionality|system|module|service|flow|workflow|pipeline|process)\b',
|
|
192
|
+
text, re.IGNORECASE
|
|
193
|
+
):
|
|
194
|
+
candidate = m.group(1).lower().strip()
|
|
195
|
+
words = candidate.split()
|
|
196
|
+
if 1 <= len(words) <= MAX_ALIAS_WORDS:
|
|
197
|
+
phrases.append(candidate)
|
|
198
|
+
|
|
199
|
+
# Filter: remove stop-word-only phrases, too short/long
|
|
200
|
+
result = []
|
|
201
|
+
for phrase in phrases:
|
|
202
|
+
words = [w for w in phrase.split() if w not in STOP_WORDS and len(w) > 1]
|
|
203
|
+
if MIN_ALIAS_WORDS <= len(words) <= MAX_ALIAS_WORDS:
|
|
204
|
+
clean = ' '.join(words)
|
|
205
|
+
if clean and clean not in result:
|
|
206
|
+
result.append(clean)
|
|
207
|
+
|
|
208
|
+
return result
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
# ── Core miner ────────────────────────────────────────────────────────────────
|
|
212
|
+
|
|
213
|
+
class SessionMiner:
|
|
214
|
+
def __init__(self, project_root: Path, verbose: bool = False):
|
|
215
|
+
self.project_root = project_root
|
|
216
|
+
self.verbose = verbose
|
|
217
|
+
# alias → {targets: set, raw_score: float, last_seen: str, count: int}
|
|
218
|
+
self._data: dict[str, dict] = {}
|
|
219
|
+
|
|
220
|
+
def mine(self, session_files: list[Path]) -> None:
|
|
221
|
+
for path in session_files:
|
|
222
|
+
self._mine_file(path)
|
|
223
|
+
|
|
224
|
+
def _mine_file(self, path: Path) -> None:
|
|
225
|
+
messages = parse_session(path)
|
|
226
|
+
if not messages:
|
|
227
|
+
return
|
|
228
|
+
|
|
229
|
+
session_date = extract_session_date(messages)
|
|
230
|
+
if session_date is None:
|
|
231
|
+
session_date = datetime.fromtimestamp(path.stat().st_mtime, tz=timezone.utc)
|
|
232
|
+
|
|
233
|
+
weight = recency_weight(session_date)
|
|
234
|
+
if weight == 0.0:
|
|
235
|
+
if self.verbose:
|
|
236
|
+
print(f" [skip] {path.name} — too old")
|
|
237
|
+
return
|
|
238
|
+
|
|
239
|
+
if self.verbose:
|
|
240
|
+
print(f" [mine] {path.name} (weight={weight}, date={session_date.date()})")
|
|
241
|
+
|
|
242
|
+
# Walk messages in pairs: look for user message followed by tool calls
|
|
243
|
+
for i, msg in enumerate(messages):
|
|
244
|
+
role = msg.get('role') or msg.get('type') or ''
|
|
245
|
+
if role not in ('user', 'human'):
|
|
246
|
+
continue
|
|
247
|
+
|
|
248
|
+
# Extract text from this user message
|
|
249
|
+
content = msg.get('content') or ''
|
|
250
|
+
if isinstance(content, list):
|
|
251
|
+
text_parts = []
|
|
252
|
+
for block in content:
|
|
253
|
+
if isinstance(block, dict) and block.get('type') == 'text':
|
|
254
|
+
text_parts.append(block.get('text', ''))
|
|
255
|
+
elif isinstance(block, str):
|
|
256
|
+
text_parts.append(block)
|
|
257
|
+
text = ' '.join(text_parts)
|
|
258
|
+
else:
|
|
259
|
+
text = str(content)
|
|
260
|
+
|
|
261
|
+
if len(text) < 5:
|
|
262
|
+
continue
|
|
263
|
+
|
|
264
|
+
# Find tool calls in subsequent assistant messages (within next 3 messages)
|
|
265
|
+
file_paths: list[str] = []
|
|
266
|
+
for j in range(i + 1, min(i + 4, len(messages))):
|
|
267
|
+
next_msg = messages[j]
|
|
268
|
+
next_role = next_msg.get('role') or next_msg.get('type') or ''
|
|
269
|
+
if next_role in ('user', 'human') and j > i + 1:
|
|
270
|
+
break # New user turn — stop looking
|
|
271
|
+
file_paths.extend(extract_file_paths_from_tool_calls([next_msg]))
|
|
272
|
+
|
|
273
|
+
if not file_paths:
|
|
274
|
+
continue
|
|
275
|
+
|
|
276
|
+
# Extract phrases from user text and pair with opened files
|
|
277
|
+
phrases = extract_user_phrases(text)
|
|
278
|
+
for phrase in phrases:
|
|
279
|
+
for fp in file_paths:
|
|
280
|
+
self._record(phrase, fp, weight, session_date)
|
|
281
|
+
|
|
282
|
+
def _record(self, alias: str, target: str, weight: float, date: datetime) -> None:
|
|
283
|
+
if alias not in self._data:
|
|
284
|
+
self._data[alias] = {
|
|
285
|
+
'targets': set(),
|
|
286
|
+
'raw_score': 0.0,
|
|
287
|
+
'last_seen': date.date().isoformat(),
|
|
288
|
+
'count': 0,
|
|
289
|
+
}
|
|
290
|
+
entry = self._data[alias]
|
|
291
|
+
entry['targets'].add(target)
|
|
292
|
+
entry['raw_score'] += weight
|
|
293
|
+
entry['count'] += 1
|
|
294
|
+
if date.date().isoformat() > entry['last_seen']:
|
|
295
|
+
entry['last_seen'] = date.date().isoformat()
|
|
296
|
+
|
|
297
|
+
def results(self) -> dict[str, dict]:
|
|
298
|
+
"""Return aliases with score >= MIN_SCORE, serializable."""
|
|
299
|
+
out = {}
|
|
300
|
+
for alias, data in self._data.items():
|
|
301
|
+
score = data['raw_score']
|
|
302
|
+
if score >= MIN_SCORE:
|
|
303
|
+
out[alias] = {
|
|
304
|
+
'targets': sorted(data['targets']),
|
|
305
|
+
'score': round(score, 2),
|
|
306
|
+
'count': data['count'],
|
|
307
|
+
'last_seen': data['last_seen'],
|
|
308
|
+
'source': 'session-mined',
|
|
309
|
+
}
|
|
310
|
+
return dict(sorted(out.items(), key=lambda x: -x[1]['score']))
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
# ── Merge with existing learned vocabulary ────────────────────────────────────
|
|
314
|
+
|
|
315
|
+
def merge_learned(existing: dict, new: dict) -> dict:
|
|
316
|
+
"""
|
|
317
|
+
Merge new mined aliases into existing learned vocabulary.
|
|
318
|
+
New entries add to score; existing entries are updated.
|
|
319
|
+
Code-derived entries (source != 'session-mined') are never overwritten.
|
|
320
|
+
"""
|
|
321
|
+
merged = dict(existing)
|
|
322
|
+
for alias, data in new.items():
|
|
323
|
+
if alias in merged:
|
|
324
|
+
existing_entry = merged[alias]
|
|
325
|
+
# Don't overwrite code-derived entries
|
|
326
|
+
if existing_entry.get('source') != 'session-mined':
|
|
327
|
+
continue
|
|
328
|
+
# Merge scores and targets
|
|
329
|
+
existing_entry['score'] = round(
|
|
330
|
+
existing_entry.get('score', 0) + data['score'], 2
|
|
331
|
+
)
|
|
332
|
+
existing_entry['count'] = existing_entry.get('count', 0) + data['count']
|
|
333
|
+
existing_entry['targets'] = sorted(
|
|
334
|
+
set(existing_entry.get('targets', [])) | set(data['targets'])
|
|
335
|
+
)
|
|
336
|
+
if data['last_seen'] > existing_entry.get('last_seen', ''):
|
|
337
|
+
existing_entry['last_seen'] = data['last_seen']
|
|
338
|
+
else:
|
|
339
|
+
merged[alias] = data
|
|
340
|
+
return merged
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def decay_old_entries(vocab: dict) -> dict:
|
|
344
|
+
"""Remove entries not seen in > 90 sessions / score below 1.0."""
|
|
345
|
+
cutoff = (datetime.now() - timedelta(days=90)).date().isoformat()
|
|
346
|
+
return {
|
|
347
|
+
alias: data
|
|
348
|
+
for alias, data in vocab.items()
|
|
349
|
+
if data.get('source') != 'session-mined'
|
|
350
|
+
or (data.get('last_seen', '0000-00-00') >= cutoff and data.get('score', 0) >= 1.0)
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
# ── Main ──────────────────────────────────────────────────────────────────────
|
|
355
|
+
|
|
356
|
+
def main() -> None:
|
|
357
|
+
parser = argparse.ArgumentParser(description='Mine Claude Code sessions for vocabulary aliases')
|
|
358
|
+
parser.add_argument('--project-root', type=Path, default=None)
|
|
359
|
+
parser.add_argument('--dry-run', action='store_true', help='Print results without saving')
|
|
360
|
+
parser.add_argument('--verbose', '-v', action='store_true', help='Show per-file details')
|
|
361
|
+
args = parser.parse_args()
|
|
362
|
+
|
|
363
|
+
global PROJECT_ROOT
|
|
364
|
+
if args.project_root:
|
|
365
|
+
PROJECT_ROOT = args.project_root.resolve()
|
|
366
|
+
|
|
367
|
+
print(f"[mine-sessions] Project root: {PROJECT_ROOT}")
|
|
368
|
+
|
|
369
|
+
# Find session files
|
|
370
|
+
session_files = find_session_files(PROJECT_ROOT)
|
|
371
|
+
if not session_files:
|
|
372
|
+
print("[mine-sessions] No session files found in ~/.claude/projects/")
|
|
373
|
+
print(f" Looked for: ~/.claude/projects/*{PROJECT_ROOT.name.lower()}*/*.jsonl")
|
|
374
|
+
print(" Sessions accumulate as you use Claude Code in this repo.")
|
|
375
|
+
|
|
376
|
+
# Still update learned-vocabulary.json to ensure it exists and is valid
|
|
377
|
+
if not LEARNED_VOC.exists():
|
|
378
|
+
LEARNED_VOC.write_text('{}', encoding='utf-8')
|
|
379
|
+
print("[mine-sessions] Created empty learned-vocabulary.json")
|
|
380
|
+
sys.exit(0)
|
|
381
|
+
|
|
382
|
+
print(f"[mine-sessions] Found {len(session_files)} session file(s)")
|
|
383
|
+
|
|
384
|
+
# Mine
|
|
385
|
+
miner = SessionMiner(PROJECT_ROOT, verbose=args.verbose)
|
|
386
|
+
miner.mine(session_files)
|
|
387
|
+
new_vocab = miner.results()
|
|
388
|
+
|
|
389
|
+
print(f"[mine-sessions] Extracted {len(new_vocab)} alias(es) scoring >= {MIN_SCORE}")
|
|
390
|
+
|
|
391
|
+
if args.verbose and new_vocab:
|
|
392
|
+
print("\n Top aliases:")
|
|
393
|
+
for alias, data in list(new_vocab.items())[:10]:
|
|
394
|
+
targets = ', '.join(data['targets'][:2])
|
|
395
|
+
print(f" {data['score']:5.1f} {alias:<30} → {targets}")
|
|
396
|
+
print()
|
|
397
|
+
|
|
398
|
+
if args.dry_run:
|
|
399
|
+
print("[mine-sessions] Dry run — not saving")
|
|
400
|
+
sys.exit(0)
|
|
401
|
+
|
|
402
|
+
# Load existing, merge, decay, save
|
|
403
|
+
existing: dict = {}
|
|
404
|
+
if LEARNED_VOC.exists():
|
|
405
|
+
try:
|
|
406
|
+
existing = json.loads(LEARNED_VOC.read_text())
|
|
407
|
+
except Exception:
|
|
408
|
+
pass
|
|
409
|
+
|
|
410
|
+
merged = merge_learned(existing, new_vocab)
|
|
411
|
+
merged = decay_old_entries(merged)
|
|
412
|
+
|
|
413
|
+
LEARNED_VOC.write_text(json.dumps(merged, indent=2, ensure_ascii=False), encoding='utf-8')
|
|
414
|
+
print(f"[mine-sessions] ✓ Saved {len(merged)} total aliases to {LEARNED_VOC}")
|
|
415
|
+
print(f" (run 'python generate.py --force' to rebuild sections with updated vocabulary)")
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
if __name__ == '__main__':
|
|
419
|
+
main()
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# Codebase Mapper — Installation Report
|
|
2
|
+
|
|
3
|
+
**Generated**: 2026-03-31 11:00:17
|
|
4
|
+
**Iteration**: 1 of 3
|
|
5
|
+
**Score**: 97.0%
|
|
6
|
+
**Result**: ✅ PASSED (90% threshold)
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Scoring Breakdown
|
|
11
|
+
|
|
12
|
+
| Category | Score | Weight | Weighted | Status |
|
|
13
|
+
|----------|-------|--------|----------|--------|
|
|
14
|
+
| Section Completeness | 100.0% | 25% | 25.0 | ✅ 19/19 sections present |
|
|
15
|
+
| Vocabulary Accuracy | 100.0% | 20% | 20.0 | ✅ 1/1 entries have valid file references |
|
|
16
|
+
| Import Chain Validity | 80.0% | 15% | 12.0 | ⚠️ No chains traced (greenfield or non-Python stack — ok) |
|
|
17
|
+
| Secret Safety | 100.0% | 15% | 15.0 | ✅ Checked 20 files — 0 potential leak(s) |
|
|
18
|
+
| Section Size Bounds | 100.0% | 10% | 10.0 | ✅ 19 sections, 0 size violations |
|
|
19
|
+
| Structural Integrity | 100.0% | 10% | 10.0 | ✅ 0 broken TOC links, 0 structural issues |
|
|
20
|
+
| Checksum Functionality | 100.0% | 5% | 5.0 | ✅ checksums.json valid (hash: e3b0c44298fc...) |
|
|
21
|
+
|
|
22
|
+
**Total: 97.0 / 100**
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## Issues Found
|
|
27
|
+
|
|
28
|
+
_No issues found._
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Final Verdict
|
|
33
|
+
|
|
34
|
+
✅ **PASSED (97.0%)** — Map is production-ready.
|
|
35
|
+
|
|
36
|
+
The project map meets quality standards and is ready to use.
|
|
37
|
+
Future sessions will load relevant sections automatically via the developer skill.
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## How to Use the Project Map
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
# Read the map index
|
|
45
|
+
cat .claude/project-map/PROJECT_MAP.md
|
|
46
|
+
|
|
47
|
+
# Force regeneration
|
|
48
|
+
python .claude/project-map/generate.py --force
|
|
49
|
+
|
|
50
|
+
# Re-run grader
|
|
51
|
+
python .claude/project-map/grader.py
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
_Report generated by `grader.py` — part of the Codebase Mapper plugin._
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# Codebase Mapper — Installation Report
|
|
2
|
+
|
|
3
|
+
**Generated**: 2026-03-31 11:00:17
|
|
4
|
+
**Iteration**: 1 of 3
|
|
5
|
+
**Score**: 97.0%
|
|
6
|
+
**Result**: ✅ PASSED (90% threshold)
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## Scoring Breakdown
|
|
11
|
+
|
|
12
|
+
| Category | Score | Weight | Weighted | Status |
|
|
13
|
+
|----------|-------|--------|----------|--------|
|
|
14
|
+
| Section Completeness | 100.0% | 25% | 25.0 | ✅ 19/19 sections present |
|
|
15
|
+
| Vocabulary Accuracy | 100.0% | 20% | 20.0 | ✅ 1/1 entries have valid file references |
|
|
16
|
+
| Import Chain Validity | 80.0% | 15% | 12.0 | ⚠️ No chains traced (greenfield or non-Python stack — ok) |
|
|
17
|
+
| Secret Safety | 100.0% | 15% | 15.0 | ✅ Checked 20 files — 0 potential leak(s) |
|
|
18
|
+
| Section Size Bounds | 100.0% | 10% | 10.0 | ✅ 19 sections, 0 size violations |
|
|
19
|
+
| Structural Integrity | 100.0% | 10% | 10.0 | ✅ 0 broken TOC links, 0 structural issues |
|
|
20
|
+
| Checksum Functionality | 100.0% | 5% | 5.0 | ✅ checksums.json valid (hash: e3b0c44298fc...) |
|
|
21
|
+
|
|
22
|
+
**Total: 97.0 / 100**
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
## Issues Found
|
|
27
|
+
|
|
28
|
+
_No issues found._
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Final Verdict
|
|
33
|
+
|
|
34
|
+
✅ **PASSED (97.0%)** — Map is production-ready.
|
|
35
|
+
|
|
36
|
+
The project map meets quality standards and is ready to use.
|
|
37
|
+
Future sessions will load relevant sections automatically via the developer skill.
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## How to Use the Project Map
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
# Read the map index
|
|
45
|
+
cat .claude/project-map/PROJECT_MAP.md
|
|
46
|
+
|
|
47
|
+
# Force regeneration
|
|
48
|
+
python .claude/project-map/generate.py --force
|
|
49
|
+
|
|
50
|
+
# Re-run grader
|
|
51
|
+
python .claude/project-map/grader.py
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
_Report generated by `grader.py` — part of the Codebase Mapper plugin._
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
# Section 10 — Tools & Commands
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
| Name | Command | Source |
|
|
5
|
+
|------|---------|--------|
|
|
6
|
+
| `postinstall` | `npm run postinstall` | package.json |
|
|
7
|
+
| `install.sh` | `bash install.sh` | shell |
|
|
8
|
+
| `/babel-fish-developer-skill` | `/babel-fish-developer-skill` | skills |
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Section 16 — Infrastructure Profile
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
| Property | Value |
|
|
5
|
+
|----------|-------|
|
|
6
|
+
| **Language** | unknown |
|
|
7
|
+
| **Framework** | unknown |
|
|
8
|
+
| **Database** | unknown |
|
|
9
|
+
| **ORM** | unknown |
|
|
10
|
+
| **Auth** | unknown |
|
|
11
|
+
| **Package Manager** | unknown |
|
|
12
|
+
| **Infrastructure** | none |
|
|
13
|
+
| **Services** | 0 docker services |
|