teamai-cli 0.25.0 → 0.26.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/README.zh-CN.md +6 -0
- package/dist/index.js +6147 -3346
- package/package.json +4 -1
- package/skill-data/core/SKILL.md +114 -0
- package/skill-data/core/references/commands.md +339 -0
- package/{skills/teamai → skill-data/core}/references/contribute-member.md +13 -10
- package/{skills/teamai → skill-data/core}/references/troubleshooting.md +9 -1
- package/skill-data/setup/SKILL.md +76 -0
- package/{skills/teamai → skill-data/setup}/references/join-member.md +17 -14
- package/{skills/teamai → skill-data/setup}/references/manage-admin.md +18 -6
- package/{skills/teamai → skill-data/setup}/references/provider-tgit.md +9 -6
- package/{skills/teamai → skill-data/setup}/references/setup-admin.md +41 -35
- package/skill-data/share/SKILL.md +70 -0
- package/skill-data/share/references/doc-template.md +44 -0
- package/skill-data/wiki/SKILL.md +314 -0
- package/skill-data/wiki/references/agents/graph-rag-agent.md +344 -0
- package/skill-data/wiki/references/agents/kb-doc-generator.md +323 -0
- package/skill-data/wiki/references/methodology/phase0-collection.md +54 -0
- package/skill-data/wiki/references/methodology/phase1-reverse-engineering.md +89 -0
- package/skill-data/wiki/references/methodology/phase2-document-types.md +341 -0
- package/skill-data/wiki/references/methodology/phase3-ai-enhancement.md +164 -0
- package/skill-data/wiki/references/methodology/phase4-quality.md +232 -0
- package/skill-data/wiki/references/overview.md +124 -0
- package/skill-data/wiki/references/phases/k1-reverse-engineering.md +118 -0
- package/skill-data/wiki/references/phases/k2-documents.md +68 -0
- package/skill-data/wiki/references/phases/k3-ai-native.md +121 -0
- package/skill-data/wiki/references/phases/k4-quality.md +190 -0
- package/skill-data/wiki/references/phases/phase0-init.md +112 -0
- package/skill-data/wiki/references/templates/project-overview.md +148 -0
- package/{skills/team-wiki-codebase → skill-data/wiki}/scripts/scan_repo.py +52 -52
- package/{skills/team-wiki-codebase → skill-data/wiki}/scripts/validate_kb.py +68 -62
- package/skills/teamai/SKILL.md +28 -128
- package/skills/team-wiki-codebase/README.md +0 -121
- package/skills/team-wiki-codebase/SKILL.md +0 -905
- package/skills/team-wiki-codebase/references/agents/graph-rag-agent.md +0 -344
- package/skills/team-wiki-codebase/references/agents/kb-doc-generator.md +0 -323
- package/skills/team-wiki-codebase/references/methodology/phase0-collection.md +0 -54
- package/skills/team-wiki-codebase/references/methodology/phase1-reverse-engineering.md +0 -89
- package/skills/team-wiki-codebase/references/methodology/phase2-document-types.md +0 -341
- package/skills/team-wiki-codebase/references/methodology/phase3-ai-enhancement.md +0 -164
- package/skills/team-wiki-codebase/references/methodology/phase4-quality.md +0 -232
- package/skills/team-wiki-codebase/references/templates/project-overview.md +0 -148
- package/skills/teamai-share-learnings/SKILL.md +0 -87
- /package/{skills/teamai → skill-data/setup}/references/uninstall.md +0 -0
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
"""
|
|
3
|
-
scan_repo.py
|
|
3
|
+
scan_repo.py: repository structure scan and statistics tool
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
1.
|
|
7
|
-
2.
|
|
8
|
-
3.
|
|
9
|
-
4.
|
|
5
|
+
Purpose: in the Phase 0 source material collection stage, quickly scan the target repository/directory and print:
|
|
6
|
+
1. Directory tree (2 levels deep)
|
|
7
|
+
2. Code statistics (language distribution, file count, total lines)
|
|
8
|
+
3. Key file discovery (entry files, config files, Proto/IDL, error code definitions)
|
|
9
|
+
4. Code hotspots (top 20 files by line count)
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
Usage:
|
|
12
12
|
python3 scan_repo.py /path/to/repo
|
|
13
13
|
python3 scan_repo.py /path/to/repo --depth 3 --top 30
|
|
14
14
|
"""
|
|
@@ -19,38 +19,38 @@ import argparse
|
|
|
19
19
|
from pathlib import Path
|
|
20
20
|
from collections import defaultdict, Counter
|
|
21
21
|
|
|
22
|
-
#
|
|
22
|
+
# Key file match patterns
|
|
23
23
|
KEY_FILE_PATTERNS = {
|
|
24
|
-
"
|
|
24
|
+
"Entry files": [
|
|
25
25
|
"main.py", "main.go", "app.py", "app.ts", "app.js",
|
|
26
26
|
"server.py", "server.go", "wsgi.py", "manage.py",
|
|
27
27
|
"cmd/*/main.go", "index.ts", "index.js",
|
|
28
28
|
],
|
|
29
|
-
"
|
|
29
|
+
"Routes/Handlers": [
|
|
30
30
|
"*handler*", "*router*", "*controller*", "*dispatch*",
|
|
31
31
|
"*route*", "*api.*", "*endpoint*",
|
|
32
32
|
],
|
|
33
|
-
"
|
|
33
|
+
"Config files": [
|
|
34
34
|
"*.yaml", "*.yml", "*.toml", "*.ini", "*.conf",
|
|
35
35
|
"*config*", "*.env", "*.env.*",
|
|
36
36
|
],
|
|
37
37
|
"Proto/IDL": [
|
|
38
38
|
"*.proto", "*.thrift", "*.graphql", "*schema*",
|
|
39
39
|
],
|
|
40
|
-
"
|
|
40
|
+
"Database/Models": [
|
|
41
41
|
"*model*", "*dao*", "*repository*", "*migration*",
|
|
42
42
|
"*schema*", "*.sql", "*db*",
|
|
43
43
|
],
|
|
44
|
-
"
|
|
44
|
+
"Constants/Error codes": [
|
|
45
45
|
"*const*", "*constant*", "*error*", "*code*",
|
|
46
46
|
"*enum*", "*define*", "*exception*",
|
|
47
47
|
],
|
|
48
|
-
"
|
|
48
|
+
"Test files": [
|
|
49
49
|
"*_test.*", "test_*", "*.spec.*", "*_spec.*",
|
|
50
50
|
],
|
|
51
51
|
}
|
|
52
52
|
|
|
53
|
-
#
|
|
53
|
+
# Language extension map
|
|
54
54
|
LANG_MAP = {
|
|
55
55
|
".py": "Python", ".go": "Go", ".js": "JavaScript", ".ts": "TypeScript",
|
|
56
56
|
".java": "Java", ".rs": "Rust", ".rb": "Ruby", ".php": "PHP",
|
|
@@ -61,7 +61,7 @@ LANG_MAP = {
|
|
|
61
61
|
".json": "JSON", ".xml": "XML", ".md": "Markdown",
|
|
62
62
|
}
|
|
63
63
|
|
|
64
|
-
#
|
|
64
|
+
# Ignored directories
|
|
65
65
|
IGNORE_DIRS = {
|
|
66
66
|
".git", ".svn", "node_modules", "__pycache__", ".tox", ".mypy_cache",
|
|
67
67
|
"venv", ".venv", "env", ".env", "vendor", "dist", "build",
|
|
@@ -85,28 +85,28 @@ def count_lines(filepath: Path) -> int:
|
|
|
85
85
|
|
|
86
86
|
|
|
87
87
|
def match_pattern(filename: str, pattern: str) -> bool:
|
|
88
|
-
"""
|
|
88
|
+
"""Simple wildcard match"""
|
|
89
89
|
import fnmatch
|
|
90
90
|
return fnmatch.fnmatch(filename.lower(), pattern.lower())
|
|
91
91
|
|
|
92
92
|
|
|
93
93
|
def scan_repository(repo_path: Path, depth: int = 2, top_n: int = 20):
|
|
94
|
-
"""
|
|
94
|
+
"""Scan the repository and return the statistics"""
|
|
95
95
|
|
|
96
96
|
all_files = []
|
|
97
|
-
lang_stats = Counter() #
|
|
97
|
+
lang_stats = Counter() # language -> (file count, line count)
|
|
98
98
|
lang_lines = Counter()
|
|
99
99
|
key_files = defaultdict(list)
|
|
100
100
|
dir_tree = []
|
|
101
101
|
|
|
102
|
-
#
|
|
102
|
+
# Walk the files
|
|
103
103
|
for root, dirs, files in os.walk(repo_path):
|
|
104
104
|
rel_root = Path(root).relative_to(repo_path)
|
|
105
105
|
|
|
106
|
-
#
|
|
106
|
+
# Skip ignored directories
|
|
107
107
|
dirs[:] = [d for d in dirs if d not in IGNORE_DIRS and not d.endswith(".egg-info")]
|
|
108
108
|
|
|
109
|
-
#
|
|
109
|
+
# Directory tree (depth-limited)
|
|
110
110
|
level = len(rel_root.parts)
|
|
111
111
|
if level <= depth:
|
|
112
112
|
indent = " " * level
|
|
@@ -124,13 +124,13 @@ def scan_repository(repo_path: Path, depth: int = 2, top_n: int = 20):
|
|
|
124
124
|
|
|
125
125
|
all_files.append((rel_path, ext, lines))
|
|
126
126
|
|
|
127
|
-
#
|
|
127
|
+
# Language statistics
|
|
128
128
|
lang = LANG_MAP.get(ext)
|
|
129
129
|
if lang:
|
|
130
130
|
lang_stats[lang] += 1
|
|
131
131
|
lang_lines[lang] += lines
|
|
132
132
|
|
|
133
|
-
#
|
|
133
|
+
# Key file matching
|
|
134
134
|
for category, patterns in KEY_FILE_PATTERNS.items():
|
|
135
135
|
for pattern in patterns:
|
|
136
136
|
if match_pattern(fname, pattern):
|
|
@@ -141,35 +141,35 @@ def scan_repository(repo_path: Path, depth: int = 2, top_n: int = 20):
|
|
|
141
141
|
|
|
142
142
|
|
|
143
143
|
def print_report(repo_path: Path, all_files, lang_stats, lang_lines, key_files, dir_tree, top_n: int):
|
|
144
|
-
"""
|
|
144
|
+
"""Print the scan report"""
|
|
145
145
|
|
|
146
146
|
total_files = len(all_files)
|
|
147
147
|
total_lines = sum(f[2] for f in all_files)
|
|
148
148
|
|
|
149
149
|
print("=" * 70)
|
|
150
|
-
print(f"
|
|
151
|
-
print(f"
|
|
150
|
+
print(f" Repository scan report: {repo_path.name}")
|
|
151
|
+
print(f" Path: {repo_path}")
|
|
152
152
|
print("=" * 70)
|
|
153
153
|
|
|
154
|
-
# 1.
|
|
155
|
-
print(f"\n## 1.
|
|
156
|
-
print(f"|
|
|
154
|
+
# 1. Basic statistics
|
|
155
|
+
print(f"\n## 1. Basic statistics\n")
|
|
156
|
+
print(f"| Metric | Value |")
|
|
157
157
|
print(f"|------|------|")
|
|
158
|
-
print(f"|
|
|
159
|
-
print(f"|
|
|
160
|
-
print(f"|
|
|
158
|
+
print(f"| Total files | {total_files} |")
|
|
159
|
+
print(f"| Total lines of code | {total_lines:,} |")
|
|
160
|
+
print(f"| Languages | {len(lang_stats)} |")
|
|
161
161
|
|
|
162
|
-
# 2.
|
|
163
|
-
print(f"\n## 2.
|
|
164
|
-
print(f"|
|
|
162
|
+
# 2. Language distribution
|
|
163
|
+
print(f"\n## 2. Language distribution\n")
|
|
164
|
+
print(f"| Language | Files | Lines | Share |")
|
|
165
165
|
print(f"|------|--------|---------|------|")
|
|
166
166
|
for lang, count in lang_stats.most_common(15):
|
|
167
167
|
lines = lang_lines[lang]
|
|
168
168
|
pct = f"{lines / total_lines * 100:.1f}%" if total_lines > 0 else "0%"
|
|
169
169
|
print(f"| {lang} | {count} | {lines:,} | {pct} |")
|
|
170
170
|
|
|
171
|
-
# 3.
|
|
172
|
-
print(f"\n## 3.
|
|
171
|
+
# 3. Directory structure
|
|
172
|
+
print(f"\n## 3. Directory structure (first 30 lines)\n")
|
|
173
173
|
print("```")
|
|
174
174
|
for line in dir_tree[:30]:
|
|
175
175
|
print(line)
|
|
@@ -177,41 +177,41 @@ def print_report(repo_path: Path, all_files, lang_stats, lang_lines, key_files,
|
|
|
177
177
|
print(f" ... ({len(dir_tree) - 30} more directories)")
|
|
178
178
|
print("```")
|
|
179
179
|
|
|
180
|
-
# 4.
|
|
181
|
-
print(f"\n## 4.
|
|
180
|
+
# 4. Key file discovery
|
|
181
|
+
print(f"\n## 4. Key file discovery\n")
|
|
182
182
|
for category, files in key_files.items():
|
|
183
183
|
if files:
|
|
184
|
-
print(f"\n### {category} ({len(files)}
|
|
185
|
-
#
|
|
184
|
+
print(f"\n### {category} ({len(files)} files)\n")
|
|
185
|
+
# Deduplicate and sort
|
|
186
186
|
seen = set()
|
|
187
187
|
for fpath, lines in sorted(files, key=lambda x: -x[1])[:10]:
|
|
188
188
|
if fpath not in seen:
|
|
189
189
|
seen.add(fpath)
|
|
190
|
-
print(f"- `{fpath}` ({lines:,}
|
|
190
|
+
print(f"- `{fpath}` ({lines:,} lines)")
|
|
191
191
|
|
|
192
|
-
# 5.
|
|
193
|
-
print(f"\n## 5.
|
|
194
|
-
print(f"|
|
|
192
|
+
# 5. Code hotspots
|
|
193
|
+
print(f"\n## 5. Code hotspots (Top {top_n})\n")
|
|
194
|
+
print(f"| Rank | File | Lines |")
|
|
195
195
|
print(f"|------|------|------|")
|
|
196
196
|
sorted_files = sorted(all_files, key=lambda x: -x[2])
|
|
197
197
|
for i, (fpath, ext, lines) in enumerate(sorted_files[:top_n], 1):
|
|
198
198
|
print(f"| {i} | `{fpath}` | {lines:,} |")
|
|
199
199
|
|
|
200
200
|
print(f"\n{'=' * 70}")
|
|
201
|
-
print(f"
|
|
201
|
+
print(f" Scan complete. {total_files} files, {total_lines:,} lines of code.")
|
|
202
202
|
print(f"{'=' * 70}")
|
|
203
203
|
|
|
204
204
|
|
|
205
205
|
def main():
|
|
206
|
-
parser = argparse.ArgumentParser(description="
|
|
207
|
-
parser.add_argument("repo_path", help="
|
|
208
|
-
parser.add_argument("--depth", type=int, default=2, help="
|
|
209
|
-
parser.add_argument("--top", type=int, default=20, help="
|
|
206
|
+
parser = argparse.ArgumentParser(description="Repository structure scan and statistics tool")
|
|
207
|
+
parser.add_argument("repo_path", help="Path of the repository/directory to scan")
|
|
208
|
+
parser.add_argument("--depth", type=int, default=2, help="Directory tree depth (default 2)")
|
|
209
|
+
parser.add_argument("--top", type=int, default=20, help="Code hotspots top N (default 20)")
|
|
210
210
|
args = parser.parse_args()
|
|
211
211
|
|
|
212
212
|
repo_path = Path(args.repo_path).resolve()
|
|
213
213
|
if not repo_path.is_dir():
|
|
214
|
-
print(f"
|
|
214
|
+
print(f"Error: {repo_path} is not a valid directory", file=sys.stderr)
|
|
215
215
|
sys.exit(1)
|
|
216
216
|
|
|
217
217
|
all_files, lang_stats, lang_lines, key_files, dir_tree = scan_repository(
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
"""
|
|
3
|
-
validate_kb.py
|
|
3
|
+
validate_kb.py: knowledge base quality validation tool
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
1.
|
|
7
|
-
2. search-anchor
|
|
8
|
-
3. AI
|
|
9
|
-
4.
|
|
10
|
-
5. README
|
|
5
|
+
Purpose: in the Phase 4 quality assessment stage, automatically check the generated knowledge base for:
|
|
6
|
+
1. Link integrity (dead link detection)
|
|
7
|
+
2. search-anchor coverage
|
|
8
|
+
3. AI Quick Reference table coverage
|
|
9
|
+
4. Bidirectional link integrity
|
|
10
|
+
5. README index coverage
|
|
11
11
|
|
|
12
|
-
|
|
12
|
+
Usage:
|
|
13
13
|
python3 validate_kb.py /path/to/knowledge-base-dir
|
|
14
14
|
python3 validate_kb.py /path/to/knowledge-base-dir --verbose
|
|
15
15
|
"""
|
|
@@ -21,18 +21,24 @@ import argparse
|
|
|
21
21
|
from pathlib import Path
|
|
22
22
|
from collections import defaultdict
|
|
23
23
|
|
|
24
|
-
# Markdown
|
|
24
|
+
# Markdown link regex: [text](path) or [text](path#anchor)
|
|
25
25
|
LINK_PATTERN = re.compile(r'\[([^\]]*)\]\(([^)]+)\)')
|
|
26
|
-
# search-anchor
|
|
26
|
+
# search-anchor regex
|
|
27
27
|
ANCHOR_PATTERN = re.compile(r'<!--\s*search-anchor\s*:(.*?)-->', re.DOTALL)
|
|
28
|
-
# AI
|
|
29
|
-
|
|
30
|
-
#
|
|
31
|
-
|
|
28
|
+
# AI Quick Reference table regex. Matches both the current English heading and the
|
|
29
|
+
# legacy Chinese heading so knowledge bases built with earlier releases still validate.
|
|
30
|
+
# Legacy knowledge bases carry the Chinese heading; matched by code point so the source stays ASCII-only.
|
|
31
|
+
AI_TABLE_PATTERN = re.compile(r'##\s*🤖\s*AI\s*(?:Quick\s*Reference|\u5feb\u901f\u7406\u89e3)', re.IGNORECASE)
|
|
32
|
+
# Bidirectional link: a link back to the main / technical architecture document.
|
|
33
|
+
# Bilingual for the same reason as AI_TABLE_PATTERN.
|
|
34
|
+
BACK_LINK_PATTERN = re.compile(
|
|
35
|
+
r'\[📘.*(?:Technical\s*Architecture|\u4e3b\u67b6\u6784|\u6280\u672f\u67b6\u6784)|Position in the overall architecture|\u5728\u6574\u4f53\u67b6\u6784\u4e2d\u7684\u4f4d\u7f6e',
|
|
36
|
+
re.IGNORECASE,
|
|
37
|
+
)
|
|
32
38
|
|
|
33
39
|
|
|
34
40
|
def find_md_files(kb_dir: Path) -> list:
|
|
35
|
-
"""
|
|
41
|
+
"""Find all .md files"""
|
|
36
42
|
md_files = []
|
|
37
43
|
for root, dirs, files in os.walk(kb_dir):
|
|
38
44
|
dirs[:] = [d for d in dirs if not d.startswith('.')]
|
|
@@ -43,27 +49,27 @@ def find_md_files(kb_dir: Path) -> list:
|
|
|
43
49
|
|
|
44
50
|
|
|
45
51
|
def check_links(md_file: Path, kb_dir: Path) -> list:
|
|
46
|
-
"""
|
|
52
|
+
"""Check that the links in the file resolve"""
|
|
47
53
|
broken = []
|
|
48
54
|
try:
|
|
49
55
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
50
56
|
except OSError:
|
|
51
|
-
return [("READ_ERROR", str(md_file), "
|
|
57
|
+
return [("READ_ERROR", str(md_file), "cannot read file")]
|
|
52
58
|
|
|
53
59
|
for match in LINK_PATTERN.finditer(content):
|
|
54
60
|
link_text = match.group(1)
|
|
55
61
|
link_target = match.group(2)
|
|
56
62
|
|
|
57
|
-
#
|
|
63
|
+
# Skip external links and anchor-only links
|
|
58
64
|
if link_target.startswith(('http://', 'https://', 'mailto:', '#')):
|
|
59
65
|
continue
|
|
60
66
|
|
|
61
|
-
#
|
|
67
|
+
# Split path and anchor
|
|
62
68
|
path_part = link_target.split('#')[0]
|
|
63
69
|
if not path_part:
|
|
64
70
|
continue
|
|
65
71
|
|
|
66
|
-
#
|
|
72
|
+
# Resolve the relative path
|
|
67
73
|
target_path = (md_file.parent / path_part).resolve()
|
|
68
74
|
if not target_path.exists():
|
|
69
75
|
rel = str(md_file.relative_to(kb_dir))
|
|
@@ -73,7 +79,7 @@ def check_links(md_file: Path, kb_dir: Path) -> list:
|
|
|
73
79
|
|
|
74
80
|
|
|
75
81
|
def check_anchor(md_file: Path) -> bool:
|
|
76
|
-
"""
|
|
82
|
+
"""Check whether the file contains a search-anchor"""
|
|
77
83
|
try:
|
|
78
84
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
79
85
|
return bool(ANCHOR_PATTERN.search(content))
|
|
@@ -82,7 +88,7 @@ def check_anchor(md_file: Path) -> bool:
|
|
|
82
88
|
|
|
83
89
|
|
|
84
90
|
def check_ai_table(md_file: Path) -> bool:
|
|
85
|
-
"""
|
|
91
|
+
"""Check whether the file contains the AI Quick Reference table"""
|
|
86
92
|
try:
|
|
87
93
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
88
94
|
return bool(AI_TABLE_PATTERN.search(content))
|
|
@@ -91,7 +97,7 @@ def check_ai_table(md_file: Path) -> bool:
|
|
|
91
97
|
|
|
92
98
|
|
|
93
99
|
def check_back_link(md_file: Path) -> bool:
|
|
94
|
-
"""
|
|
100
|
+
"""Check whether the component document links back to the main architecture document"""
|
|
95
101
|
try:
|
|
96
102
|
content = md_file.read_text(encoding='utf-8', errors='ignore')
|
|
97
103
|
return bool(BACK_LINK_PATTERN.search(content))
|
|
@@ -100,7 +106,7 @@ def check_back_link(md_file: Path) -> bool:
|
|
|
100
106
|
|
|
101
107
|
|
|
102
108
|
def check_readme_coverage(kb_dir: Path, md_files: list) -> tuple:
|
|
103
|
-
"""
|
|
109
|
+
"""Check whether the README indexes every .md file"""
|
|
104
110
|
readme_path = kb_dir / "README.md"
|
|
105
111
|
if not readme_path.exists():
|
|
106
112
|
return [], md_files
|
|
@@ -112,7 +118,7 @@ def check_readme_coverage(kb_dir: Path, md_files: list) -> tuple:
|
|
|
112
118
|
for f in md_files:
|
|
113
119
|
if f.name == "README.md":
|
|
114
120
|
continue
|
|
115
|
-
#
|
|
121
|
+
# Check whether the README mentions this file
|
|
116
122
|
fname_no_ext = f.stem
|
|
117
123
|
if fname_no_ext in readme_content or f.name in readme_content:
|
|
118
124
|
covered.append(f)
|
|
@@ -123,106 +129,106 @@ def check_readme_coverage(kb_dir: Path, md_files: list) -> tuple:
|
|
|
123
129
|
|
|
124
130
|
|
|
125
131
|
def main():
|
|
126
|
-
parser = argparse.ArgumentParser(description="
|
|
127
|
-
parser.add_argument("kb_dir", help="
|
|
128
|
-
parser.add_argument("--verbose", "-v", action="store_true", help="
|
|
132
|
+
parser = argparse.ArgumentParser(description="Knowledge base quality validation tool")
|
|
133
|
+
parser.add_argument("kb_dir", help="Path of the knowledge base directory")
|
|
134
|
+
parser.add_argument("--verbose", "-v", action="store_true", help="Print details")
|
|
129
135
|
args = parser.parse_args()
|
|
130
136
|
|
|
131
137
|
kb_dir = Path(args.kb_dir).resolve()
|
|
132
138
|
if not kb_dir.is_dir():
|
|
133
|
-
print(f"
|
|
139
|
+
print(f"Error: {kb_dir} is not a valid directory", file=sys.stderr)
|
|
134
140
|
sys.exit(1)
|
|
135
141
|
|
|
136
142
|
md_files = find_md_files(kb_dir)
|
|
137
143
|
if not md_files:
|
|
138
|
-
print(f"
|
|
144
|
+
print(f"Warning: no .md files found in {kb_dir}")
|
|
139
145
|
sys.exit(0)
|
|
140
146
|
|
|
141
|
-
#
|
|
147
|
+
# Filter the component design documents (files starting with a number)
|
|
142
148
|
component_docs = [f for f in md_files if re.match(r'^\d+_', f.name)]
|
|
143
149
|
|
|
144
150
|
print("=" * 70)
|
|
145
|
-
print(f"
|
|
146
|
-
print(f"
|
|
147
|
-
print(f"
|
|
151
|
+
print(f" Knowledge base quality validation report")
|
|
152
|
+
print(f" Directory: {kb_dir}")
|
|
153
|
+
print(f" Files: {len(md_files)} .md files ({len(component_docs)} component documents)")
|
|
148
154
|
print("=" * 70)
|
|
149
155
|
|
|
150
156
|
total_score = 0
|
|
151
157
|
max_score = 0
|
|
152
158
|
|
|
153
|
-
# 1.
|
|
154
|
-
print(f"\n## 1.
|
|
159
|
+
# 1. Link integrity
|
|
160
|
+
print(f"\n## 1. Link integrity check\n")
|
|
155
161
|
all_broken = []
|
|
156
162
|
for f in md_files:
|
|
157
163
|
broken = check_links(f, kb_dir)
|
|
158
164
|
all_broken.extend(broken)
|
|
159
165
|
|
|
160
166
|
if all_broken:
|
|
161
|
-
print(f"❌
|
|
167
|
+
print(f"❌ Found {len(all_broken)} dead links:")
|
|
162
168
|
for src, target, text in all_broken[:20]:
|
|
163
169
|
print(f" {src} → [{text}]({target})")
|
|
164
170
|
if len(all_broken) > 20:
|
|
165
|
-
print(f" ...
|
|
171
|
+
print(f" ... and {len(all_broken) - 20} more")
|
|
166
172
|
else:
|
|
167
|
-
print(f"✅
|
|
173
|
+
print(f"✅ All links valid ({len(md_files)} files checked)")
|
|
168
174
|
total_score += 20
|
|
169
175
|
max_score += 20
|
|
170
176
|
|
|
171
|
-
# 2. search-anchor
|
|
172
|
-
print(f"\n## 2. Search-Anchor
|
|
177
|
+
# 2. search-anchor coverage
|
|
178
|
+
print(f"\n## 2. Search-Anchor coverage\n")
|
|
173
179
|
has_anchor = sum(1 for f in md_files if check_anchor(f))
|
|
174
180
|
anchor_pct = has_anchor / len(md_files) * 100 if md_files else 0
|
|
175
|
-
print(f"{'✅' if anchor_pct >= 80 else '⚠️'} {has_anchor}/{len(md_files)}
|
|
181
|
+
print(f"{'✅' if anchor_pct >= 80 else '⚠️'} {has_anchor}/{len(md_files)} files have a search-anchor ({anchor_pct:.0f}%)")
|
|
176
182
|
if args.verbose:
|
|
177
183
|
for f in md_files:
|
|
178
184
|
if not check_anchor(f):
|
|
179
|
-
print(f"
|
|
185
|
+
print(f" Missing: {f.relative_to(kb_dir)}")
|
|
180
186
|
if anchor_pct >= 80:
|
|
181
187
|
total_score += 20
|
|
182
188
|
elif anchor_pct >= 50:
|
|
183
189
|
total_score += 10
|
|
184
190
|
max_score += 20
|
|
185
191
|
|
|
186
|
-
# 3. AI
|
|
187
|
-
print(f"\n## 3. AI
|
|
192
|
+
# 3. AI Quick Reference table coverage (component documents only)
|
|
193
|
+
print(f"\n## 3. AI Quick Reference table coverage (component documents)\n")
|
|
188
194
|
if component_docs:
|
|
189
195
|
has_ai_table = sum(1 for f in component_docs if check_ai_table(f))
|
|
190
196
|
ai_pct = has_ai_table / len(component_docs) * 100
|
|
191
|
-
print(f"{'✅' if ai_pct >= 90 else '⚠️'} {has_ai_table}/{len(component_docs)}
|
|
197
|
+
print(f"{'✅' if ai_pct >= 90 else '⚠️'} {has_ai_table}/{len(component_docs)} component documents have an AI Quick Reference table ({ai_pct:.0f}%)")
|
|
192
198
|
if args.verbose:
|
|
193
199
|
for f in component_docs:
|
|
194
200
|
if not check_ai_table(f):
|
|
195
|
-
print(f"
|
|
201
|
+
print(f" Missing: {f.relative_to(kb_dir)}")
|
|
196
202
|
if ai_pct >= 90:
|
|
197
203
|
total_score += 20
|
|
198
204
|
elif ai_pct >= 60:
|
|
199
205
|
total_score += 10
|
|
200
206
|
else:
|
|
201
|
-
print("⚠️
|
|
207
|
+
print("⚠️ No numbered component documents found")
|
|
202
208
|
max_score += 20
|
|
203
209
|
|
|
204
|
-
# 4.
|
|
205
|
-
print(f"\n## 4.
|
|
210
|
+
# 4. Bidirectional link check (component documents link back to the main architecture)
|
|
211
|
+
print(f"\n## 4. Bidirectional link check (component → main architecture)\n")
|
|
206
212
|
if component_docs:
|
|
207
213
|
has_back = sum(1 for f in component_docs if check_back_link(f))
|
|
208
214
|
back_pct = has_back / len(component_docs) * 100
|
|
209
|
-
print(f"{'✅' if back_pct >= 90 else '⚠️'} {has_back}/{len(component_docs)}
|
|
215
|
+
print(f"{'✅' if back_pct >= 90 else '⚠️'} {has_back}/{len(component_docs)} component documents link back to the main architecture ({back_pct:.0f}%)")
|
|
210
216
|
if back_pct >= 90:
|
|
211
217
|
total_score += 20
|
|
212
218
|
elif back_pct >= 60:
|
|
213
219
|
total_score += 10
|
|
214
220
|
else:
|
|
215
|
-
print("⚠️
|
|
221
|
+
print("⚠️ No numbered component documents found")
|
|
216
222
|
max_score += 20
|
|
217
223
|
|
|
218
|
-
# 5. README
|
|
219
|
-
print(f"\n## 5. README
|
|
224
|
+
# 5. README index coverage
|
|
225
|
+
print(f"\n## 5. README index coverage\n")
|
|
220
226
|
covered, uncovered = check_readme_coverage(kb_dir, md_files)
|
|
221
227
|
if (kb_dir / "README.md").exists():
|
|
222
228
|
cover_pct = len(covered) / (len(covered) + len(uncovered)) * 100 if (covered or uncovered) else 100
|
|
223
|
-
print(f"{'✅' if cover_pct >= 90 else '⚠️'} README
|
|
229
|
+
print(f"{'✅' if cover_pct >= 90 else '⚠️'} README indexes {len(covered)}/{len(covered)+len(uncovered)} documents ({cover_pct:.0f}%)")
|
|
224
230
|
if uncovered and args.verbose:
|
|
225
|
-
print("
|
|
231
|
+
print(" Not indexed:")
|
|
226
232
|
for f in uncovered[:10]:
|
|
227
233
|
print(f" {f.relative_to(kb_dir)}")
|
|
228
234
|
if cover_pct >= 90:
|
|
@@ -230,19 +236,19 @@ def main():
|
|
|
230
236
|
elif cover_pct >= 60:
|
|
231
237
|
total_score += 10
|
|
232
238
|
else:
|
|
233
|
-
print("❌
|
|
239
|
+
print("❌ README.md not found")
|
|
234
240
|
max_score += 20
|
|
235
241
|
|
|
236
|
-
#
|
|
242
|
+
# Summary
|
|
237
243
|
final_pct = total_score / max_score * 100 if max_score else 0
|
|
238
244
|
print(f"\n{'=' * 70}")
|
|
239
|
-
print(f"
|
|
245
|
+
print(f" Overall score: {total_score}/{max_score} ({final_pct:.0f}%)")
|
|
240
246
|
if final_pct >= 90:
|
|
241
|
-
print(f"
|
|
247
|
+
print(f" Rating: ✅ Excellent. The knowledge base meets the quality bar")
|
|
242
248
|
elif final_pct >= 70:
|
|
243
|
-
print(f"
|
|
249
|
+
print(f" Rating: ⚠️ Good. Fixing the issues above is recommended")
|
|
244
250
|
else:
|
|
245
|
-
print(f"
|
|
251
|
+
print(f" Rating: ❌ Needs improvement. There are many quality issues")
|
|
246
252
|
print(f"{'=' * 70}")
|
|
247
253
|
|
|
248
254
|
|