skillvariants 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ """skillvariants spike package."""
2
+
3
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from .cli import app
2
+
3
+ app()
@@ -0,0 +1,185 @@
1
+ """Deterministic mutation classification between a target skill and a candidate.
2
+
3
+ Every label comes with human-readable evidence so nothing is asserted without
4
+ a traceable reason (spec section 14, risk E).
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import re
9
+ from dataclasses import dataclass, field
10
+
11
+ from .features import (
12
+ PROJECT_PHRASES,
13
+ SkillFeatures,
14
+ uppercase_rules,
15
+ )
16
+ from .parser import SkillDoc
17
+ from .similarity import ScoreBreakdown, copy_labels
18
+
19
+ # Display/canonical priority when several labels apply.
20
+ LABEL_PRIORITY: tuple[str, ...] = (
21
+ "exact-copy",
22
+ "body-copy-with-metadata-change",
23
+ "compatibility-wrapper",
24
+ "routing-specialization",
25
+ "compact-rewrite",
26
+ "expanded-guidance",
27
+ "workflow-specialization",
28
+ "project-specialization",
29
+ )
30
+
31
+ PROJECT_PHRASE_RE = re.compile(
32
+ r"\b(" + "|".join(re.escape(p) for p in PROJECT_PHRASES) + r")\b",
33
+ re.IGNORECASE,
34
+ )
35
+
36
+
37
+ @dataclass
38
+ class Classification:
39
+ labels: list[str] = field(default_factory=list)
40
+ evidence: list[str] = field(default_factory=list)
41
+
42
+ @property
43
+ def primary(self) -> str:
44
+ return self.labels[0] if self.labels else "no-label"
45
+
46
+ def as_dict(self) -> dict:
47
+ return {"labels": self.labels, "primary": self.primary, "evidence": self.evidence}
48
+
49
+
50
+ def classify_pair(
51
+ target: SkillDoc,
52
+ target_feats: SkillFeatures,
53
+ candidate: SkillDoc,
54
+ candidate_feats: SkillFeatures,
55
+ sim: ScoreBreakdown,
56
+ ) -> Classification:
57
+ labels: list[str] = []
58
+ evidence: list[str] = []
59
+
60
+ target_len = max(len(target.body), 1)
61
+ ratio = len(candidate.body) / target_len
62
+
63
+ copies = copy_labels(target, candidate)
64
+ if copies:
65
+ labels.extend(copies)
66
+ if copies[0] == "exact-copy":
67
+ evidence.append("normalized full-file SHA-256 hashes are equal")
68
+ else:
69
+ evidence.append("normalized body hashes equal; frontmatter differs")
70
+
71
+ same_name = sim.name_match
72
+
73
+ if candidate_feats.is_wrapper and same_name:
74
+ labels.append("compatibility-wrapper")
75
+ evidence.append(
76
+ f"short body ({candidate_feats.n_lines} lines) with canonical reference"
77
+ + (f" -> {candidate_feats.canonical_ref}" if candidate_feats.canonical_ref else "")
78
+ )
79
+
80
+ if same_name and not copies:
81
+ if (
82
+ 0.0 < ratio <= 0.5
83
+ and 0.30 <= sim.token_set_ratio < 0.60
84
+ and not candidate_feats.is_wrapper
85
+ ):
86
+ labels.append("compact-rewrite")
87
+ evidence.append(
88
+ f"body length {ratio:.0%} of reference; token similarity "
89
+ f"{sim.token_set_ratio:.0%}"
90
+ )
91
+ elif ratio >= 1.3 and sim.token_set_ratio >= 0.45:
92
+ labels.append("expanded-guidance")
93
+ evidence.append(
94
+ f"body length {ratio:.0%} of reference; token similarity "
95
+ f"{sim.token_set_ratio:.0%}"
96
+ )
97
+
98
+ new_routing = set(candidate_feats.routing_signals) - set(
99
+ target_feats.routing_signals
100
+ )
101
+ new_refs = set(candidate_feats.cross_skill_refs) - set(
102
+ target_feats.cross_skill_refs
103
+ )
104
+ if (
105
+ new_routing
106
+ and sim.token_set_ratio >= 0.30
107
+ and (len(new_routing) >= 2 or len(new_refs) >= 1)
108
+ ):
109
+ labels.append("routing-specialization")
110
+ evidence.append(
111
+ f"new routing signals: {sorted(new_routing)[:4]}; "
112
+ f"new skill references: {sorted(new_refs)[:4] or 'none'}"
113
+ )
114
+
115
+ # Tightened in spike 2 (spec section 13): workflow-specialization must
116
+ # show real workflow-structure evidence -- a genuine turnover of named
117
+ # sections -- instead of firing on generic textual drift.
118
+ headings_removed_count = len(
119
+ {h.lower() for h in target_feats.headings}
120
+ - {h.lower() for h in candidate_feats.headings}
121
+ )
122
+ headings_added_count = len(
123
+ {h.lower() for h in candidate_feats.headings}
124
+ - {h.lower() for h in target_feats.headings}
125
+ )
126
+ if (
127
+ sim.token_set_ratio >= 0.30
128
+ and headings_removed_count + headings_added_count >= 3
129
+ and sim.heading_jaccard < 0.70
130
+ and not candidate_feats.is_wrapper
131
+ ):
132
+ labels.append("workflow-specialization")
133
+ evidence.append(
134
+ f"workflow structure reworked: "
135
+ f"{headings_removed_count} headings removed, "
136
+ f"{headings_added_count} added"
137
+ f" ({len(target_feats.headings)} -> {len(candidate_feats.headings)})"
138
+ )
139
+
140
+ project_signals = PROJECT_PHRASE_RE.findall(candidate.body)
141
+ if len(set(project_signals)) >= 2:
142
+ labels.append("project-specialization")
143
+ evidence.append(f"project-specific phrases: {sorted(set(project_signals))}")
144
+
145
+ ordered = [label for label in LABEL_PRIORITY if label in labels]
146
+ return Classification(labels=ordered, evidence=evidence)
147
+
148
+
149
+ def mutation_summary(
150
+ target: SkillDoc,
151
+ target_feats: SkillFeatures,
152
+ candidate: SkillDoc,
153
+ candidate_feats: SkillFeatures,
154
+ sim: ScoreBreakdown,
155
+ classification: Classification,
156
+ ) -> dict:
157
+ """Deterministic, human-readable mutation summary (spec section 15)."""
158
+ target_len = max(len(target.body), 1)
159
+ ratio = len(candidate.body) / target_len
160
+ target_headings = {h.lower() for h in target_feats.headings}
161
+ candidate_headings = {h.lower() for h in candidate_feats.headings}
162
+
163
+ target_rules = set(uppercase_rules(target.body))
164
+ candidate_rules = set(uppercase_rules(candidate.body))
165
+
166
+ delta = (len(candidate.body) - len(target.body)) / target_len
167
+ return {
168
+ "type": classification.primary,
169
+ "labels": classification.labels,
170
+ "length_change": f"{delta:+.0%}",
171
+ "workflow_headings": (
172
+ f"{len(target_feats.headings)} -> {len(candidate_feats.headings)} headings"
173
+ ),
174
+ "preserved_rules": sorted(target_rules & candidate_rules),
175
+ "added_rules": sorted(candidate_rules - target_rules),
176
+ "removed_rules": sorted(target_rules - candidate_rules),
177
+ "added_headings": sorted(candidate_headings - target_headings),
178
+ "removed_headings": sorted(target_headings - candidate_headings),
179
+ "code_blocks": (target_feats.n_code_blocks, candidate_feats.n_code_blocks),
180
+ "cross_skill_refs": (
181
+ len(target_feats.cross_skill_refs),
182
+ len(candidate_feats.cross_skill_refs),
183
+ ),
184
+ "commands": (len(target_feats.commands), len(candidate_feats.commands)),
185
+ }
skillvariants/cli.py ADDED
@@ -0,0 +1,379 @@
1
+ """skillvariants CLI: inspect / related / compare (archetype-first mutation explorer).
2
+
3
+ `related` exposes two ranking modes:
4
+ --mode mutations (default) archetype map of notable adaptation patterns
5
+ --mode closest pure similarity DESC after exact-copy collapse
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import sys
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ import typer
15
+ from rich.console import Console
16
+
17
+ from .classify import classify_pair
18
+ from .features import SkillFeatures, extract_features
19
+ from .github import GitHubClient, GitHubError
20
+ from .parser import GitHubRef, parse_github_url, parse_skill_md
21
+ from .ranking import (
22
+ MIN_RELATEDNESS,
23
+ ARCHETYPE_HUMAN_LABELS,
24
+ ArchetypeBucket,
25
+ build_archetype_map,
26
+ build_variant_row,
27
+ representative_score,
28
+ _archetype_signals,
29
+ )
30
+ from .render import SimilarityRow, render_compare, render_inspect, render_mutations, render_related
31
+ from .similarity import normalize_for_hash, score_similarity, sha256
32
+
33
+ app = typer.Typer(
34
+ help="SkillVariants: find copies and variants of Agent Skills across GitHub.",
35
+ no_args_is_help=True,
36
+ )
37
+ console = Console()
38
+
39
+ DEFAULT_CACHE_DIR = ".cache/skillvariants"
40
+ DEFAULT_MAX_PAGES = 3
41
+
42
+
43
+ def _representative_view(group, archetype: str) -> tuple[float, list[str]]:
44
+ """Score + signals for a group's representative under one archetype."""
45
+ return (
46
+ representative_score(group.representative, archetype),
47
+ _archetype_signals(group.representative, archetype),
48
+ )
49
+
50
+
51
+ def emit_json(data: dict | list) -> None:
52
+ """All JSON output goes through here: never through Rich markup."""
53
+ sys.stdout.write(json.dumps(data, indent=2, ensure_ascii=False) + "\n")
54
+
55
+
56
+ def _load_doc(
57
+ client: Any, url: str
58
+ ) -> tuple[GitHubRef, Any, SkillFeatures]:
59
+ ref = parse_github_url(url)
60
+ doc = parse_skill_md(client.fetch_text(ref), source=ref)
61
+ return ref, doc, extract_features(doc)
62
+
63
+
64
+ def _exit_usage(exc: Exception) -> None:
65
+ if isinstance(exc, ValueError):
66
+ raise typer.BadParameter(str(exc))
67
+ raise typer.Exit(str(exc), code=1)
68
+
69
+
70
+ @app.command()
71
+ def inspect(
72
+ url: str = typer.Argument(..., help="GitHub SKILL.md URL"),
73
+ cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
74
+ json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
75
+ ) -> None:
76
+ """Show frontmatter, body stats, and signals for one Skill."""
77
+ try:
78
+ client = GitHubClient(cache_dir=cache_dir)
79
+ ref, doc, feats = _load_doc(client, url)
80
+ except (ValueError, GitHubError) as exc:
81
+ _exit_usage(exc)
82
+ if json_output:
83
+ emit_json(
84
+ {
85
+ "ref": {
86
+ "owner": ref.owner,
87
+ "repo": ref.repo,
88
+ "ref": ref.ref,
89
+ "path": ref.path,
90
+ },
91
+ "name": doc.name,
92
+ "frontmatter": doc.frontmatter,
93
+ "body": {
94
+ "lines": feats.n_lines,
95
+ "characters": feats.n_chars,
96
+ "headings": feats.headings,
97
+ "code_blocks": feats.n_code_blocks,
98
+ "tables": feats.n_tables,
99
+ "bullets": feats.n_bullets,
100
+ },
101
+ "signals": {
102
+ "commands": feats.commands,
103
+ "urls": feats.urls,
104
+ "cross_skill_refs": feats.cross_skill_refs,
105
+ "routing_signals": feats.routing_signals,
106
+ "wrapper_signals": feats.wrapper_signals,
107
+ "canonical_ref": feats.canonical_ref,
108
+ "is_wrapper": feats.is_wrapper,
109
+ },
110
+ }
111
+ )
112
+ else:
113
+ render_inspect(ref, doc, feats, console)
114
+
115
+
116
+ def _build_variant_pool(
117
+ target_url: str,
118
+ cache_dir: Path,
119
+ max_pages: int,
120
+ client: Any | None = None,
121
+ ) -> dict:
122
+ """Shared retrieval stage: fetch candidates, collapse exact copies,
123
+ score every unique variant against the target."""
124
+ if client is None:
125
+ client = GitHubClient(cache_dir=cache_dir)
126
+ target_ref, target, target_feats = _load_doc(client, target_url)
127
+ name = target.name
128
+ if not name:
129
+ raise ValueError(
130
+ "The target Skill has no `name` in its frontmatter; cannot search "
131
+ "by name. Rerun with a different SKILL.md URL."
132
+ )
133
+ query = f'"name: {name}" filename:SKILL.md'
134
+ hits = client.code_search(query, max_pages=max_pages)
135
+
136
+ seen: set[tuple[str, str]] = set()
137
+ candidates = []
138
+ for hit in hits:
139
+ key = (hit.repo, hit.path)
140
+ # NOTE: an empty default_branch is legitimate -- the contents API
141
+ # treats a blank ?ref= as the repository default branch (cached
142
+ # first-spike searches rely on this).
143
+ if key in seen or hit.repo == "":
144
+ continue
145
+ seen.add(key)
146
+ if hit.repo == target_ref.repo_slug and hit.path == target_ref.path:
147
+ continue # the target itself
148
+ candidates.append(hit)
149
+
150
+ target_hash = sha256(normalize_for_hash(target.raw))
151
+ fetch_errors: list[str] = []
152
+ exact_copies_of_target = 0
153
+ by_hash: dict[str, VariantRow] = {}
154
+ fetched_count = 0
155
+
156
+ for hit in candidates:
157
+ try:
158
+ text = client.fetch_text(hit.to_ref())
159
+ except GitHubError as exc:
160
+ fetch_errors.append(f"{hit.repo}/{hit.path}: {exc}")
161
+ continue
162
+ except FileNotFoundError as exc: # only reachable via test fakes
163
+ fetch_errors.append(f"{hit.repo}/{hit.path}: missing fixture {exc}")
164
+ continue
165
+ fetched_count += 1
166
+ doc = parse_skill_md(text, source=hit.to_ref())
167
+ full_hash = sha256(normalize_for_hash(doc.raw))
168
+ if full_hash == target_hash:
169
+ exact_copies_of_target += 1
170
+ if full_hash in by_hash:
171
+ by_hash[full_hash].copy_count += 1
172
+ continue
173
+ feats = extract_features(doc)
174
+ sim = score_similarity(target, target_feats, doc, feats)
175
+ classification = classify_pair(target, target_feats, doc, feats, sim)
176
+ row = build_variant_row(
177
+ repo=hit.repo,
178
+ path=hit.path,
179
+ ref=hit.default_branch,
180
+ doc=doc,
181
+ feats=feats,
182
+ sim=sim,
183
+ classification=classification,
184
+ copy_count=1,
185
+ sha256_full=full_hash,
186
+ target_doc=target,
187
+ target_feats=target_feats,
188
+ target_name=name,
189
+ )
190
+ by_hash[full_hash] = row
191
+
192
+ pool = sorted(by_hash.values(), key=lambda r: r.sim.score, reverse=True)
193
+ return {
194
+ "target": {"repo": target_ref.repo_slug, "path": target_ref.path, "name": name},
195
+ "query": query,
196
+ "counts": {
197
+ "candidates_fetched": fetched_count,
198
+ "candidates_total": len(candidates),
199
+ "exact_copies_of_target": exact_copies_of_target,
200
+ "unique_variants": len(pool),
201
+ },
202
+ "fetch_errors": fetch_errors,
203
+ "pool": pool,
204
+ "target_doc": target,
205
+ }
206
+
207
+
208
+ def _payload_common(pool_data: dict, mode: str) -> dict:
209
+ return {
210
+ "target": pool_data["target"],
211
+ "query": pool_data["query"],
212
+ "mode": mode,
213
+ "counts": pool_data["counts"],
214
+ "fetch_errors": pool_data["fetch_errors"],
215
+ }
216
+
217
+
218
+ def _print_fetch_errors(pool_data: dict) -> None:
219
+ if pool_data["fetch_errors"]:
220
+ for error in pool_data["fetch_errors"][:5]:
221
+ console.print(f"[yellow]SKIPPED[/yellow] {error}")
222
+
223
+
224
+ def _build_mutations_payload(pool_data: dict, limit: int) -> dict:
225
+ """Assemble the archetype-first mutations view (JSON schema, section 16)."""
226
+ counts = pool_data["counts"]
227
+ pool = pool_data["pool"]
228
+ buckets, summary_counts = build_archetype_map(
229
+ pool, representatives_per_archetype=limit
230
+ )
231
+
232
+ def archetype_payload(bucket: ArchetypeBucket) -> dict:
233
+ representatives = []
234
+ for group in bucket.ranked_groups[:limit]:
235
+ rep = group.representative
236
+ score, signals = _representative_view(group, bucket.archetype)
237
+ representatives.append(
238
+ {
239
+ "repository": rep.repo,
240
+ "path": rep.path,
241
+ "ref": rep.ref,
242
+ "sha256_full": rep.sha256_full,
243
+ "relatedness_score": round(rep.relatedness, 4),
244
+ "representative_score": round(score, 4),
245
+ "group_member_count": len(group.members),
246
+ "group_occurrence_count": group.member_count,
247
+ "similarity_percent": round(rep.sim.score * 100),
248
+ "primary_mutation_type": rep.classification.primary,
249
+ "labels": rep.classification.labels,
250
+ "signals": signals,
251
+ "description": rep.doc.description,
252
+ "body_excerpt": " ".join(rep.doc.body.split())[:400],
253
+ }
254
+ )
255
+ return {
256
+ "type": bucket.archetype,
257
+ "label": ARCHETYPE_HUMAN_LABELS.get(bucket.archetype, bucket.archetype),
258
+ "group_count": len(bucket.ranked_groups),
259
+ "unique_variant_count": bucket.unique_variant_count,
260
+ "occurrence_count": bucket.occurrence_count,
261
+ "representatives": representatives,
262
+ }
263
+
264
+ return {
265
+ "skill": {
266
+ "name": pool_data["target"]["name"],
267
+ "repository": pool_data["target"]["repo"],
268
+ "path": pool_data["target"]["path"],
269
+ },
270
+ "query": pool_data["query"],
271
+ "mode": "mutations",
272
+ "exact_copy_count": counts["exact_copies_of_target"],
273
+ "unique_related_variants": counts["unique_variants"],
274
+ "counts": {**counts, **summary_counts},
275
+ "min_relatedness": MIN_RELATEDNESS,
276
+ "fetch_errors": pool_data["fetch_errors"],
277
+ "archetypes": [archetype_payload(b) for b in buckets],
278
+ }
279
+
280
+
281
+ @app.command()
282
+ def related(
283
+ url: str = typer.Argument(..., help="GitHub SKILL.md URL"),
284
+ mode: str = typer.Option(
285
+ "mutations",
286
+ "--mode",
287
+ help="Ranking view: 'mutations' (archetype map, default) or 'closest'.",
288
+ ),
289
+ limit: int = typer.Option(10, "--limit", min=1, max=50),
290
+ max_pages: int = typer.Option(DEFAULT_MAX_PAGES, "--max-pages", min=1, max=10),
291
+ cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
292
+ json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
293
+ ) -> None:
294
+ """Find same-name variants: notable mutations (default) or closest copies."""
295
+ if mode not in ("mutations", "closest"):
296
+ raise typer.BadParameter("--mode must be 'mutations' or 'closest'")
297
+ try:
298
+ pool_data = _build_variant_pool(url, cache_dir, max_pages)
299
+ except (ValueError, GitHubError) as exc:
300
+ _exit_usage(exc)
301
+
302
+ counts = pool_data["counts"]
303
+ if mode == "closest":
304
+ payload = _payload_common(pool_data, "closest")
305
+ payload["rows"] = [row.summary() for row in pool_data["pool"][:limit]]
306
+ if json_output:
307
+ emit_json(payload)
308
+ return
309
+ rows = [
310
+ SimilarityRow(
311
+ rank=index,
312
+ repo=row["repo"],
313
+ path=row["path"],
314
+ score=row["similarity_score"],
315
+ label=row["label"]
316
+ + (f" (x{row['copy_count']})" if row["copy_count"] > 1 else ""),
317
+ description=row["description"] or "",
318
+ copy_count=row["copy_count"],
319
+ )
320
+ for index, row in enumerate(payload["rows"], start=1)
321
+ ]
322
+ render_related(
323
+ family_name=pool_data["target"]["name"] or "Unknown",
324
+ total_candidates=counts["candidates_total"],
325
+ rows=rows,
326
+ exact_copies=counts["exact_copies_of_target"],
327
+ unique_variants=min(limit, counts["unique_variants"]),
328
+ console=console,
329
+ )
330
+ _print_fetch_errors(pool_data)
331
+ return
332
+
333
+ # ---- mutations mode (default): archetype-first view ------------------
334
+ payload = _build_mutations_payload(pool_data, limit)
335
+ if json_output:
336
+ emit_json(payload)
337
+ return
338
+ render_mutations(
339
+ skill_name=pool_data["target"]["name"] or "Unknown",
340
+ total_candidates=counts["candidates_total"],
341
+ payload=payload,
342
+ console=console,
343
+ )
344
+ _print_fetch_errors(pool_data)
345
+
346
+
347
+ @app.command()
348
+ def compare(
349
+ url_a: str = typer.Argument(..., help="First GitHub SKILL.md URL"),
350
+ url_b: str = typer.Argument(..., help="Second GitHub SKILL.md URL"),
351
+ cache_dir: Path = typer.Option(DEFAULT_CACHE_DIR, "--cache-dir"),
352
+ json_output: bool = typer.Option(False, "--json", help="Emit machine-readable JSON"),
353
+ ) -> None:
354
+ """Compare two Skills and classify the mutation between them."""
355
+ try:
356
+ client = GitHubClient(cache_dir=cache_dir)
357
+ ref_a, doc_a, feats_a = _load_doc(client, url_a)
358
+ ref_b, doc_b, feats_b = _load_doc(client, url_b)
359
+ sim = score_similarity(doc_a, feats_a, doc_b, feats_b)
360
+ classification = classify_pair(doc_a, feats_a, doc_b, feats_b, sim)
361
+ except (ValueError, GitHubError) as exc:
362
+ _exit_usage(exc)
363
+ if json_output:
364
+ emit_json(
365
+ {
366
+ "a": ref_a.slug,
367
+ "b": ref_b.slug,
368
+ "similarity": sim.as_dict(),
369
+ "classification": classification.as_dict(),
370
+ }
371
+ )
372
+ else:
373
+ render_compare(
374
+ ref_a, doc_a, feats_a, ref_b, doc_b, feats_b, sim, classification, console
375
+ )
376
+
377
+
378
+ if __name__ == "__main__":
379
+ app()