skillvariants 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- skillvariants/__init__.py +3 -0
- skillvariants/__main__.py +3 -0
- skillvariants/classify.py +185 -0
- skillvariants/cli.py +379 -0
- skillvariants/features.py +208 -0
- skillvariants/github.py +183 -0
- skillvariants/parser.py +176 -0
- skillvariants/ranking.py +804 -0
- skillvariants/render.py +255 -0
- skillvariants/similarity.py +134 -0
- skillvariants-0.1.1.dist-info/METADATA +219 -0
- skillvariants-0.1.1.dist-info/RECORD +16 -0
- skillvariants-0.1.1.dist-info/WHEEL +5 -0
- skillvariants-0.1.1.dist-info/entry_points.txt +2 -0
- skillvariants-0.1.1.dist-info/licenses/LICENSE +202 -0
- skillvariants-0.1.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Deterministic mutation classification between a target skill and a candidate.
|
|
2
|
+
|
|
3
|
+
Every label comes with human-readable evidence so nothing is asserted without
|
|
4
|
+
a traceable reason (spec section 14, risk E).
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
|
|
11
|
+
from .features import (
|
|
12
|
+
PROJECT_PHRASES,
|
|
13
|
+
SkillFeatures,
|
|
14
|
+
uppercase_rules,
|
|
15
|
+
)
|
|
16
|
+
from .parser import SkillDoc
|
|
17
|
+
from .similarity import ScoreBreakdown, copy_labels
|
|
18
|
+
|
|
19
|
+
# Display/canonical priority when several labels apply.
|
|
20
|
+
LABEL_PRIORITY: tuple[str, ...] = (
|
|
21
|
+
"exact-copy",
|
|
22
|
+
"body-copy-with-metadata-change",
|
|
23
|
+
"compatibility-wrapper",
|
|
24
|
+
"routing-specialization",
|
|
25
|
+
"compact-rewrite",
|
|
26
|
+
"expanded-guidance",
|
|
27
|
+
"workflow-specialization",
|
|
28
|
+
"project-specialization",
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
PROJECT_PHRASE_RE = re.compile(
|
|
32
|
+
r"\b(" + "|".join(re.escape(p) for p in PROJECT_PHRASES) + r")\b",
|
|
33
|
+
re.IGNORECASE,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class Classification:
|
|
39
|
+
labels: list[str] = field(default_factory=list)
|
|
40
|
+
evidence: list[str] = field(default_factory=list)
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def primary(self) -> str:
|
|
44
|
+
return self.labels[0] if self.labels else "no-label"
|
|
45
|
+
|
|
46
|
+
def as_dict(self) -> dict:
|
|
47
|
+
return {"labels": self.labels, "primary": self.primary, "evidence": self.evidence}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def classify_pair(
|
|
51
|
+
target: SkillDoc,
|
|
52
|
+
target_feats: SkillFeatures,
|
|
53
|
+
candidate: SkillDoc,
|
|
54
|
+
candidate_feats: SkillFeatures,
|
|
55
|
+
sim: ScoreBreakdown,
|
|
56
|
+
) -> Classification:
|
|
57
|
+
labels: list[str] = []
|
|
58
|
+
evidence: list[str] = []
|
|
59
|
+
|
|
60
|
+
target_len = max(len(target.body), 1)
|
|
61
|
+
ratio = len(candidate.body) / target_len
|
|
62
|
+
|
|
63
|
+
copies = copy_labels(target, candidate)
|
|
64
|
+
if copies:
|
|
65
|
+
labels.extend(copies)
|
|
66
|
+
if copies[0] == "exact-copy":
|
|
67
|
+
evidence.append("normalized full-file SHA-256 hashes are equal")
|
|
68
|
+
else:
|
|
69
|
+
evidence.append("normalized body hashes equal; frontmatter differs")
|
|
70
|
+
|
|
71
|
+
same_name = sim.name_match
|
|
72
|
+
|
|
73
|
+
if candidate_feats.is_wrapper and same_name:
|
|
74
|
+
labels.append("compatibility-wrapper")
|
|
75
|
+
evidence.append(
|
|
76
|
+
f"short body ({candidate_feats.n_lines} lines) with canonical reference"
|
|
77
|
+
+ (f" -> {candidate_feats.canonical_ref}" if candidate_feats.canonical_ref else "")
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
if same_name and not copies:
|
|
81
|
+
if (
|
|
82
|
+
0.0 < ratio <= 0.5
|
|
83
|
+
and 0.30 <= sim.token_set_ratio < 0.60
|
|
84
|
+
and not candidate_feats.is_wrapper
|
|
85
|
+
):
|
|
86
|
+
labels.append("compact-rewrite")
|
|
87
|
+
evidence.append(
|
|
88
|
+
f"body length {ratio:.0%} of reference; token similarity "
|
|
89
|
+
f"{sim.token_set_ratio:.0%}"
|
|
90
|
+
)
|
|
91
|
+
elif ratio >= 1.3 and sim.token_set_ratio >= 0.45:
|
|
92
|
+
labels.append("expanded-guidance")
|
|
93
|
+
evidence.append(
|
|
94
|
+
f"body length {ratio:.0%} of reference; token similarity "
|
|
95
|
+
f"{sim.token_set_ratio:.0%}"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
new_routing = set(candidate_feats.routing_signals) - set(
|
|
99
|
+
target_feats.routing_signals
|
|
100
|
+
)
|
|
101
|
+
new_refs = set(candidate_feats.cross_skill_refs) - set(
|
|
102
|
+
target_feats.cross_skill_refs
|
|
103
|
+
)
|
|
104
|
+
if (
|
|
105
|
+
new_routing
|
|
106
|
+
and sim.token_set_ratio >= 0.30
|
|
107
|
+
and (len(new_routing) >= 2 or len(new_refs) >= 1)
|
|
108
|
+
):
|
|
109
|
+
labels.append("routing-specialization")
|
|
110
|
+
evidence.append(
|
|
111
|
+
f"new routing signals: {sorted(new_routing)[:4]}; "
|
|
112
|
+
f"new skill references: {sorted(new_refs)[:4] or 'none'}"
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# Tightened in spike 2 (spec section 13): workflow-specialization must
|
|
116
|
+
# show real workflow-structure evidence -- a genuine turnover of named
|
|
117
|
+
# sections -- instead of firing on generic textual drift.
|
|
118
|
+
headings_removed_count = len(
|
|
119
|
+
{h.lower() for h in target_feats.headings}
|
|
120
|
+
- {h.lower() for h in candidate_feats.headings}
|
|
121
|
+
)
|
|
122
|
+
headings_added_count = len(
|
|
123
|
+
{h.lower() for h in candidate_feats.headings}
|
|
124
|
+
- {h.lower() for h in target_feats.headings}
|
|
125
|
+
)
|
|
126
|
+
if (
|
|
127
|
+
sim.token_set_ratio >= 0.30
|
|
128
|
+
and headings_removed_count + headings_added_count >= 3
|
|
129
|
+
and sim.heading_jaccard < 0.70
|
|
130
|
+
and not candidate_feats.is_wrapper
|
|
131
|
+
):
|
|
132
|
+
labels.append("workflow-specialization")
|
|
133
|
+
evidence.append(
|
|
134
|
+
f"workflow structure reworked: "
|
|
135
|
+
f"{headings_removed_count} headings removed, "
|
|
136
|
+
f"{headings_added_count} added"
|
|
137
|
+
f" ({len(target_feats.headings)} -> {len(candidate_feats.headings)})"
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
project_signals = PROJECT_PHRASE_RE.findall(candidate.body)
|
|
141
|
+
if len(set(project_signals)) >= 2:
|
|
142
|
+
labels.append("project-specialization")
|
|
143
|
+
evidence.append(f"project-specific phrases: {sorted(set(project_signals))}")
|
|
144
|
+
|
|
145
|
+
ordered = [label for label in LABEL_PRIORITY if label in labels]
|
|
146
|
+
return Classification(labels=ordered, evidence=evidence)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def mutation_summary(
|
|
150
|
+
target: SkillDoc,
|
|
151
|
+
target_feats: SkillFeatures,
|
|
152
|
+
candidate: SkillDoc,
|
|
153
|
+
candidate_feats: SkillFeatures,
|
|
154
|
+
sim: ScoreBreakdown,
|
|
155
|
+
classification: Classification,
|
|
156
|
+
) -> dict:
|
|
157
|
+
"""Deterministic, human-readable mutation summary (spec section 15)."""
|
|
158
|
+
target_len = max(len(target.body), 1)
|
|
159
|
+
ratio = len(candidate.body) / target_len
|
|
160
|
+
target_headings = {h.lower() for h in target_feats.headings}
|
|
161
|
+
candidate_headings = {h.lower() for h in candidate_feats.headings}
|
|
162
|
+
|
|
163
|
+
target_rules = set(uppercase_rules(target.body))
|
|
164
|
+
candidate_rules = set(uppercase_rules(candidate.body))
|
|
165
|
+
|
|
166
|
+
delta = (len(candidate.body) - len(target.body)) / target_len
|
|
167
|
+
return {
|
|
168
|
+
"type": classification.primary,
|
|
169
|
+
"labels": classification.labels,
|
|
170
|
+
"length_change": f"{delta:+.0%}",
|
|
171
|
+
"workflow_headings": (
|
|
172
|
+
f"{len(target_feats.headings)} -> {len(candidate_feats.headings)} headings"
|
|
173
|
+
),
|
|
174
|
+
"preserved_rules": sorted(target_rules & candidate_rules),
|
|
175
|
+
"added_rules": sorted(candidate_rules - target_rules),
|
|
176
|
+
"removed_rules": sorted(target_rules - candidate_rules),
|
|
177
|
+
"added_headings": sorted(candidate_headings - target_headings),
|
|
178
|
+
"removed_headings": sorted(target_headings - candidate_headings),
|
|
179
|
+
"code_blocks": (target_feats.n_code_blocks, candidate_feats.n_code_blocks),
|
|
180
|
+
"cross_skill_refs": (
|
|
181
|
+
len(target_feats.cross_skill_refs),
|
|
182
|
+
len(candidate_feats.cross_skill_refs),
|
|
183
|
+
),
|
|
184
|
+
"commands": (len(target_feats.commands), len(candidate_feats.commands)),
|
|
185
|
+
}
|
skillvariants/cli.py
ADDED
|
@@ -0,0 +1,379 @@
|
|
|
1
|
+
"""skillvariants CLI: inspect / related / compare (archetype-first mutation explorer).
|
|
2
|
+
|
|
3
|
+
`related` exposes two ranking modes:
|
|
4
|
+
--mode mutations (default) archetype map of notable adaptation patterns
|
|
5
|
+
--mode closest pure similarity DESC after exact-copy collapse
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
import typer
|
|
15
|
+
from rich.console import Console
|
|
16
|
+
|
|
17
|
+
from .classify import classify_pair
|
|
18
|
+
from .features import SkillFeatures, extract_features
|
|
19
|
+
from .github import GitHubClient, GitHubError
|
|
20
|
+
from .parser import GitHubRef, parse_github_url, parse_skill_md
|
|
21
|
+
from .ranking import (
|
|
22
|
+
MIN_RELATEDNESS,
|
|
23
|
+
ARCHETYPE_HUMAN_LABELS,
|
|
24
|
+
ArchetypeBucket,
|
|
25
|
+
build_archetype_map,
|
|
26
|
+
build_variant_row,
|
|
27
|
+
representative_score,
|
|
28
|
+
_archetype_signals,
|
|
29
|
+
)
|
|
30
|
+
from .render import SimilarityRow, render_compare, render_inspect, render_mutations, render_related
|
|
31
|
+
from .similarity import normalize_for_hash, score_similarity, sha256
|
|
32
|
+
|
|
33
|
+
app = typer.Typer(
|
|
34
|
+
help="SkillVariants: find copies and variants of Agent Skills across GitHub.",
|
|
35
|
+
no_args_is_help=True,
|
|
36
|
+
)
|
|
37
|
+
console = Console()
|
|
38
|
+
|
|
39
|
+
DEFAULT_CACHE_DIR = ".cache/skillvariants"
|
|
40
|
+
DEFAULT_MAX_PAGES = 3
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _representative_view(group, archetype: str) -> tuple[float, list[str]]:
|
|
44
|
+
"""Score + signals for a group's representative under one archetype."""
|
|
45
|
+
return (
|
|
46
|
+
representative_score(group.representative, archetype),
|
|
47
|
+
_archetype_signals(group.representative, archetype),
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def emit_json(data: dict | list) -> None:
|
|
52
|
+
"""All JSON output goes through here: never through Rich markup."""
|
|
53
|
+
sys.stdout.write(json.dumps(data, indent=2, ensure_ascii=False) + "\n")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _load_doc(
|
|
57
|
+
client: Any, url: str
|
|
58
|
+
) -> tuple[GitHubRef, Any, SkillFeatures]:
|
|
59
|
+
ref = parse_github_url(url)
|
|
60
|
+
doc = parse_skill_md(client.fetch_text(ref), source=ref)
|
|
61
|
+
return ref, doc, extract_features(doc)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _exit_usage(exc: Exception) -> None:
|
|
65
|
+
if isinstance(exc, ValueError):
|
|
66
|
+
raise typer.BadParameter(str(exc))
|
|
67
|
+
raise typer.Exit(str(exc), code=1)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@app.command()
|
|
71
|
+
def inspect(
|
|
72
|
+
url: str = typer.Argument(..., help="GitHub SKILL.md URL"),
|
|
73
|
+
cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
|
|
74
|
+
json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
|
|
75
|
+
) -> None:
|
|
76
|
+
"""Show frontmatter, body stats, and signals for one Skill."""
|
|
77
|
+
try:
|
|
78
|
+
client = GitHubClient(cache_dir=cache_dir)
|
|
79
|
+
ref, doc, feats = _load_doc(client, url)
|
|
80
|
+
except (ValueError, GitHubError) as exc:
|
|
81
|
+
_exit_usage(exc)
|
|
82
|
+
if json_output:
|
|
83
|
+
emit_json(
|
|
84
|
+
{
|
|
85
|
+
"ref": {
|
|
86
|
+
"owner": ref.owner,
|
|
87
|
+
"repo": ref.repo,
|
|
88
|
+
"ref": ref.ref,
|
|
89
|
+
"path": ref.path,
|
|
90
|
+
},
|
|
91
|
+
"name": doc.name,
|
|
92
|
+
"frontmatter": doc.frontmatter,
|
|
93
|
+
"body": {
|
|
94
|
+
"lines": feats.n_lines,
|
|
95
|
+
"characters": feats.n_chars,
|
|
96
|
+
"headings": feats.headings,
|
|
97
|
+
"code_blocks": feats.n_code_blocks,
|
|
98
|
+
"tables": feats.n_tables,
|
|
99
|
+
"bullets": feats.n_bullets,
|
|
100
|
+
},
|
|
101
|
+
"signals": {
|
|
102
|
+
"commands": feats.commands,
|
|
103
|
+
"urls": feats.urls,
|
|
104
|
+
"cross_skill_refs": feats.cross_skill_refs,
|
|
105
|
+
"routing_signals": feats.routing_signals,
|
|
106
|
+
"wrapper_signals": feats.wrapper_signals,
|
|
107
|
+
"canonical_ref": feats.canonical_ref,
|
|
108
|
+
"is_wrapper": feats.is_wrapper,
|
|
109
|
+
},
|
|
110
|
+
}
|
|
111
|
+
)
|
|
112
|
+
else:
|
|
113
|
+
render_inspect(ref, doc, feats, console)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _build_variant_pool(
|
|
117
|
+
target_url: str,
|
|
118
|
+
cache_dir: Path,
|
|
119
|
+
max_pages: int,
|
|
120
|
+
client: Any | None = None,
|
|
121
|
+
) -> dict:
|
|
122
|
+
"""Shared retrieval stage: fetch candidates, collapse exact copies,
|
|
123
|
+
score every unique variant against the target."""
|
|
124
|
+
if client is None:
|
|
125
|
+
client = GitHubClient(cache_dir=cache_dir)
|
|
126
|
+
target_ref, target, target_feats = _load_doc(client, target_url)
|
|
127
|
+
name = target.name
|
|
128
|
+
if not name:
|
|
129
|
+
raise ValueError(
|
|
130
|
+
"The target Skill has no `name` in its frontmatter; cannot search "
|
|
131
|
+
"by name. Rerun with a different SKILL.md URL."
|
|
132
|
+
)
|
|
133
|
+
query = f'"name: {name}" filename:SKILL.md'
|
|
134
|
+
hits = client.code_search(query, max_pages=max_pages)
|
|
135
|
+
|
|
136
|
+
seen: set[tuple[str, str]] = set()
|
|
137
|
+
candidates = []
|
|
138
|
+
for hit in hits:
|
|
139
|
+
key = (hit.repo, hit.path)
|
|
140
|
+
# NOTE: an empty default_branch is legitimate -- the contents API
|
|
141
|
+
# treats a blank ?ref= as the repository default branch (cached
|
|
142
|
+
# first-spike searches rely on this).
|
|
143
|
+
if key in seen or hit.repo == "":
|
|
144
|
+
continue
|
|
145
|
+
seen.add(key)
|
|
146
|
+
if hit.repo == target_ref.repo_slug and hit.path == target_ref.path:
|
|
147
|
+
continue # the target itself
|
|
148
|
+
candidates.append(hit)
|
|
149
|
+
|
|
150
|
+
target_hash = sha256(normalize_for_hash(target.raw))
|
|
151
|
+
fetch_errors: list[str] = []
|
|
152
|
+
exact_copies_of_target = 0
|
|
153
|
+
by_hash: dict[str, VariantRow] = {}
|
|
154
|
+
fetched_count = 0
|
|
155
|
+
|
|
156
|
+
for hit in candidates:
|
|
157
|
+
try:
|
|
158
|
+
text = client.fetch_text(hit.to_ref())
|
|
159
|
+
except GitHubError as exc:
|
|
160
|
+
fetch_errors.append(f"{hit.repo}/{hit.path}: {exc}")
|
|
161
|
+
continue
|
|
162
|
+
except FileNotFoundError as exc: # only reachable via test fakes
|
|
163
|
+
fetch_errors.append(f"{hit.repo}/{hit.path}: missing fixture {exc}")
|
|
164
|
+
continue
|
|
165
|
+
fetched_count += 1
|
|
166
|
+
doc = parse_skill_md(text, source=hit.to_ref())
|
|
167
|
+
full_hash = sha256(normalize_for_hash(doc.raw))
|
|
168
|
+
if full_hash == target_hash:
|
|
169
|
+
exact_copies_of_target += 1
|
|
170
|
+
if full_hash in by_hash:
|
|
171
|
+
by_hash[full_hash].copy_count += 1
|
|
172
|
+
continue
|
|
173
|
+
feats = extract_features(doc)
|
|
174
|
+
sim = score_similarity(target, target_feats, doc, feats)
|
|
175
|
+
classification = classify_pair(target, target_feats, doc, feats, sim)
|
|
176
|
+
row = build_variant_row(
|
|
177
|
+
repo=hit.repo,
|
|
178
|
+
path=hit.path,
|
|
179
|
+
ref=hit.default_branch,
|
|
180
|
+
doc=doc,
|
|
181
|
+
feats=feats,
|
|
182
|
+
sim=sim,
|
|
183
|
+
classification=classification,
|
|
184
|
+
copy_count=1,
|
|
185
|
+
sha256_full=full_hash,
|
|
186
|
+
target_doc=target,
|
|
187
|
+
target_feats=target_feats,
|
|
188
|
+
target_name=name,
|
|
189
|
+
)
|
|
190
|
+
by_hash[full_hash] = row
|
|
191
|
+
|
|
192
|
+
pool = sorted(by_hash.values(), key=lambda r: r.sim.score, reverse=True)
|
|
193
|
+
return {
|
|
194
|
+
"target": {"repo": target_ref.repo_slug, "path": target_ref.path, "name": name},
|
|
195
|
+
"query": query,
|
|
196
|
+
"counts": {
|
|
197
|
+
"candidates_fetched": fetched_count,
|
|
198
|
+
"candidates_total": len(candidates),
|
|
199
|
+
"exact_copies_of_target": exact_copies_of_target,
|
|
200
|
+
"unique_variants": len(pool),
|
|
201
|
+
},
|
|
202
|
+
"fetch_errors": fetch_errors,
|
|
203
|
+
"pool": pool,
|
|
204
|
+
"target_doc": target,
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _payload_common(pool_data: dict, mode: str) -> dict:
|
|
209
|
+
return {
|
|
210
|
+
"target": pool_data["target"],
|
|
211
|
+
"query": pool_data["query"],
|
|
212
|
+
"mode": mode,
|
|
213
|
+
"counts": pool_data["counts"],
|
|
214
|
+
"fetch_errors": pool_data["fetch_errors"],
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _print_fetch_errors(pool_data: dict) -> None:
|
|
219
|
+
if pool_data["fetch_errors"]:
|
|
220
|
+
for error in pool_data["fetch_errors"][:5]:
|
|
221
|
+
console.print(f"[yellow]SKIPPED[/yellow] {error}")
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _build_mutations_payload(pool_data: dict, limit: int) -> dict:
|
|
225
|
+
"""Assemble the archetype-first mutations view (JSON schema, section 16)."""
|
|
226
|
+
counts = pool_data["counts"]
|
|
227
|
+
pool = pool_data["pool"]
|
|
228
|
+
buckets, summary_counts = build_archetype_map(
|
|
229
|
+
pool, representatives_per_archetype=limit
|
|
230
|
+
)
|
|
231
|
+
|
|
232
|
+
def archetype_payload(bucket: ArchetypeBucket) -> dict:
|
|
233
|
+
representatives = []
|
|
234
|
+
for group in bucket.ranked_groups[:limit]:
|
|
235
|
+
rep = group.representative
|
|
236
|
+
score, signals = _representative_view(group, bucket.archetype)
|
|
237
|
+
representatives.append(
|
|
238
|
+
{
|
|
239
|
+
"repository": rep.repo,
|
|
240
|
+
"path": rep.path,
|
|
241
|
+
"ref": rep.ref,
|
|
242
|
+
"sha256_full": rep.sha256_full,
|
|
243
|
+
"relatedness_score": round(rep.relatedness, 4),
|
|
244
|
+
"representative_score": round(score, 4),
|
|
245
|
+
"group_member_count": len(group.members),
|
|
246
|
+
"group_occurrence_count": group.member_count,
|
|
247
|
+
"similarity_percent": round(rep.sim.score * 100),
|
|
248
|
+
"primary_mutation_type": rep.classification.primary,
|
|
249
|
+
"labels": rep.classification.labels,
|
|
250
|
+
"signals": signals,
|
|
251
|
+
"description": rep.doc.description,
|
|
252
|
+
"body_excerpt": " ".join(rep.doc.body.split())[:400],
|
|
253
|
+
}
|
|
254
|
+
)
|
|
255
|
+
return {
|
|
256
|
+
"type": bucket.archetype,
|
|
257
|
+
"label": ARCHETYPE_HUMAN_LABELS.get(bucket.archetype, bucket.archetype),
|
|
258
|
+
"group_count": len(bucket.ranked_groups),
|
|
259
|
+
"unique_variant_count": bucket.unique_variant_count,
|
|
260
|
+
"occurrence_count": bucket.occurrence_count,
|
|
261
|
+
"representatives": representatives,
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
return {
|
|
265
|
+
"skill": {
|
|
266
|
+
"name": pool_data["target"]["name"],
|
|
267
|
+
"repository": pool_data["target"]["repo"],
|
|
268
|
+
"path": pool_data["target"]["path"],
|
|
269
|
+
},
|
|
270
|
+
"query": pool_data["query"],
|
|
271
|
+
"mode": "mutations",
|
|
272
|
+
"exact_copy_count": counts["exact_copies_of_target"],
|
|
273
|
+
"unique_related_variants": counts["unique_variants"],
|
|
274
|
+
"counts": {**counts, **summary_counts},
|
|
275
|
+
"min_relatedness": MIN_RELATEDNESS,
|
|
276
|
+
"fetch_errors": pool_data["fetch_errors"],
|
|
277
|
+
"archetypes": [archetype_payload(b) for b in buckets],
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
@app.command()
|
|
282
|
+
def related(
|
|
283
|
+
url: str = typer.Argument(..., help="GitHub SKILL.md URL"),
|
|
284
|
+
mode: str = typer.Option(
|
|
285
|
+
"mutations",
|
|
286
|
+
"--mode",
|
|
287
|
+
help="Ranking view: 'mutations' (archetype map, default) or 'closest'.",
|
|
288
|
+
),
|
|
289
|
+
limit: int = typer.Option(10, "--limit", min=1, max=50),
|
|
290
|
+
max_pages: int = typer.Option(DEFAULT_MAX_PAGES, "--max-pages", min=1, max=10),
|
|
291
|
+
cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
|
|
292
|
+
json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
|
|
293
|
+
) -> None:
|
|
294
|
+
"""Find same-name variants: notable mutations (default) or closest copies."""
|
|
295
|
+
if mode not in ("mutations", "closest"):
|
|
296
|
+
raise typer.BadParameter("--mode must be 'mutations' or 'closest'")
|
|
297
|
+
try:
|
|
298
|
+
pool_data = _build_variant_pool(url, cache_dir, max_pages)
|
|
299
|
+
except (ValueError, GitHubError) as exc:
|
|
300
|
+
_exit_usage(exc)
|
|
301
|
+
|
|
302
|
+
counts = pool_data["counts"]
|
|
303
|
+
if mode == "closest":
|
|
304
|
+
payload = _payload_common(pool_data, "closest")
|
|
305
|
+
payload["rows"] = [row.summary() for row in pool_data["pool"][:limit]]
|
|
306
|
+
if json_output:
|
|
307
|
+
emit_json(payload)
|
|
308
|
+
return
|
|
309
|
+
rows = [
|
|
310
|
+
SimilarityRow(
|
|
311
|
+
rank=index,
|
|
312
|
+
repo=row["repo"],
|
|
313
|
+
path=row["path"],
|
|
314
|
+
score=row["similarity_score"],
|
|
315
|
+
label=row["label"]
|
|
316
|
+
+ (f" (x{row['copy_count']})" if row["copy_count"] > 1 else ""),
|
|
317
|
+
description=row["description"] or "",
|
|
318
|
+
copy_count=row["copy_count"],
|
|
319
|
+
)
|
|
320
|
+
for index, row in enumerate(payload["rows"], start=1)
|
|
321
|
+
]
|
|
322
|
+
render_related(
|
|
323
|
+
family_name=pool_data["target"]["name"] or "Unknown",
|
|
324
|
+
total_candidates=counts["candidates_total"],
|
|
325
|
+
rows=rows,
|
|
326
|
+
exact_copies=counts["exact_copies_of_target"],
|
|
327
|
+
unique_variants=min(limit, counts["unique_variants"]),
|
|
328
|
+
console=console,
|
|
329
|
+
)
|
|
330
|
+
_print_fetch_errors(pool_data)
|
|
331
|
+
return
|
|
332
|
+
|
|
333
|
+
# ---- mutations mode (default): archetype-first view ------------------
|
|
334
|
+
payload = _build_mutations_payload(pool_data, limit)
|
|
335
|
+
if json_output:
|
|
336
|
+
emit_json(payload)
|
|
337
|
+
return
|
|
338
|
+
render_mutations(
|
|
339
|
+
skill_name=pool_data["target"]["name"] or "Unknown",
|
|
340
|
+
total_candidates=counts["candidates_total"],
|
|
341
|
+
payload=payload,
|
|
342
|
+
console=console,
|
|
343
|
+
)
|
|
344
|
+
_print_fetch_errors(pool_data)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
@app.command()
|
|
348
|
+
def compare(
|
|
349
|
+
url_a: str = typer.Argument(..., help="First GitHub SKILL.md URL"),
|
|
350
|
+
url_b: str = typer.Argument(..., help="Second GitHub SKILL.md URL"),
|
|
351
|
+
cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
|
|
352
|
+
json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
|
|
353
|
+
) -> None:
|
|
354
|
+
"""Compare two Skills and classify the mutation between them."""
|
|
355
|
+
try:
|
|
356
|
+
client = GitHubClient(cache_dir=cache_dir)
|
|
357
|
+
ref_a, doc_a, feats_a = _load_doc(client, url_a)
|
|
358
|
+
ref_b, doc_b, feats_b = _load_doc(client, url_b)
|
|
359
|
+
sim = score_similarity(doc_a, feats_a, doc_b, feats_b)
|
|
360
|
+
classification = classify_pair(doc_a, feats_a, doc_b, feats_b, sim)
|
|
361
|
+
except (ValueError, GitHubError) as exc:
|
|
362
|
+
_exit_usage(exc)
|
|
363
|
+
if json_output:
|
|
364
|
+
emit_json(
|
|
365
|
+
{
|
|
366
|
+
"a": ref_a.slug,
|
|
367
|
+
"b": ref_b.slug,
|
|
368
|
+
"similarity": sim.as_dict(),
|
|
369
|
+
"classification": classification.as_dict(),
|
|
370
|
+
}
|
|
371
|
+
)
|
|
372
|
+
else:
|
|
373
|
+
render_compare(
|
|
374
|
+
ref_a, doc_a, feats_a, ref_b, doc_b, feats_b, sim, classification, console
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
if __name__ == "__main__":
|
|
379
|
+
app()
|