diffprompt 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffprompt-0.2.0/.github/workflows/ci.yml +29 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/.gitignore +12 -10
- {diffprompt-0.1.0 → diffprompt-0.2.0}/API_REFERENCE.md +2 -1
- {diffprompt-0.1.0 → diffprompt-0.2.0}/PKG-INFO +9 -1
- {diffprompt-0.1.0 → diffprompt-0.2.0}/README.md +8 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/__init__.py +2 -2
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/cli.py +20 -13
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/clusterer.py +129 -129
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/embedder.py +62 -62
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/generator.py +125 -122
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/judge.py +25 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/ontology.py +76 -24
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/runner.py +84 -65
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/slicer.py +152 -152
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/models/__init__.py +2 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/models/cascade.py +8 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/output/exporter.py +270 -226
- diffprompt-0.2.0/diffprompt/output/insights.py +142 -0
- diffprompt-0.2.0/diffprompt/output/terminal.py +188 -0
- diffprompt-0.2.0/diffprompt/ratelimit.py +50 -0
- diffprompt-0.2.0/docs/superpowers/plans/2026-06-21-diffprompt-v3-security-scanner.md +1533 -0
- diffprompt-0.2.0/docs/superpowers/specs/2026-06-19-diffprompt-v2-program-design.md +241 -0
- diffprompt-0.2.0/docs/superpowers/specs/2026-06-21-diffprompt-v3-design.md +224 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/pyproject.toml +57 -57
- {diffprompt-0.1.0 → diffprompt-0.2.0}/tests/test_generator.py +17 -9
- diffprompt-0.2.0/tests/test_insights.py +104 -0
- diffprompt-0.2.0/tests/test_judge.py +33 -0
- diffprompt-0.2.0/tests/test_ontology.py +77 -0
- diffprompt-0.2.0/tests/test_ratelimit.py +38 -0
- diffprompt-0.2.0/tests/test_runner.py +43 -0
- diffprompt-0.2.0/tests/test_scorer.py +67 -0
- diffprompt-0.1.0/diffprompt/output/terminal.py +0 -154
- {diffprompt-0.1.0 → diffprompt-0.2.0}/CONTRIBUTING.md +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/COOKBOOK.md +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/__init__.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/core/scorer.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/diffprompt/output/__init__.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/examples/basic_diff.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/examples/ontology_inspect.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/examples/similarity_playground.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/report.html +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/tests/__init__.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/tests/test_embedder.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/tests/test_slicer.py +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/v1.txt +0 -0
- {diffprompt-0.1.0 → diffprompt-0.2.0}/v2.txt +0 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
python-version: ["3.10", "3.11", "3.12"]
|
|
15
|
+
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
|
|
19
|
+
- name: Set up Python ${{ matrix.python-version }}
|
|
20
|
+
uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: ${{ matrix.python-version }}
|
|
23
|
+
cache: pip
|
|
24
|
+
|
|
25
|
+
- name: Install
|
|
26
|
+
run: pip install -e ".[dev]"
|
|
27
|
+
|
|
28
|
+
- name: Run tests
|
|
29
|
+
run: pytest tests/ -v
|
|
@@ -1,10 +1,12 @@
|
|
|
1
|
-
__pycache__/
|
|
2
|
-
*.py[cod]
|
|
3
|
-
*.egg-info/
|
|
4
|
-
dist/
|
|
5
|
-
build/
|
|
6
|
-
.env
|
|
7
|
-
.venv/
|
|
8
|
-
venv/
|
|
9
|
-
diffprompt_report.*
|
|
10
|
-
diffprompt.ontology.json
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*.egg-info/
|
|
4
|
+
dist/
|
|
5
|
+
build/
|
|
6
|
+
.env
|
|
7
|
+
.venv/
|
|
8
|
+
venv/
|
|
9
|
+
diffprompt_report.*
|
|
10
|
+
diffprompt.ontology.json
|
|
11
|
+
.diffprompt_cache/
|
|
12
|
+
.superpowers/
|
|
@@ -299,7 +299,7 @@ Generates `n` test cases distributed across the four taxonomy buckets (45% typic
|
|
|
299
299
|
#### diversity_score
|
|
300
300
|
|
|
301
301
|
```python
|
|
302
|
-
def diversity_score(test_cases: list[TestCase]
|
|
302
|
+
def diversity_score(test_cases: list[TestCase]) -> float
|
|
303
303
|
```
|
|
304
304
|
|
|
305
305
|
Computes how diverse the test suite is. Returns `1 - mean_pairwise_similarity`. Higher is more diverse. A score below 0.4 means many inputs are semantically redundant.
|
|
@@ -485,6 +485,7 @@ where `surprise = divergence * (1 - input_length / 50)`. Short inputs that chang
|
|
|
485
485
|
```python
|
|
486
486
|
async def select_key_examples(
|
|
487
487
|
diffs: list[DiffResult],
|
|
488
|
+
top_n: int = 3,
|
|
488
489
|
local_only: bool = False,
|
|
489
490
|
) -> list[KeyExample]
|
|
490
491
|
```
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: diffprompt
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: git diff for your prompt's behavior
|
|
5
5
|
Project-URL: Homepage, https://github.com/RudraDudhat2509/diffprompt
|
|
6
6
|
Project-URL: Repository, https://github.com/RudraDudhat2509/diffprompt
|
|
@@ -30,6 +30,8 @@ Description-Content-Type: text/markdown
|
|
|
30
30
|
|
|
31
31
|
# diffprompt
|
|
32
32
|
|
|
33
|
+
[](https://github.com/RudraDudhat2509/diffprompt/actions/workflows/ci.yml)
|
|
34
|
+
|
|
33
35
|
> git diff for your prompt's behavior
|
|
34
36
|
|
|
35
37
|
```bash
|
|
@@ -161,6 +163,12 @@ Different job. Different tool.
|
|
|
161
163
|
|
|
162
164
|
Override any layer with `--model` and `--judge`.
|
|
163
165
|
|
|
166
|
+
Groq calls are paced to stay under the free-tier rate limit. Tune it with `DIFFPROMPT_GROQ_RPM` (default 30; set `0` to disable):
|
|
167
|
+
|
|
168
|
+
```bash
|
|
169
|
+
export DIFFPROMPT_GROQ_RPM=60 # raise on paid tiers
|
|
170
|
+
```
|
|
171
|
+
|
|
164
172
|
---
|
|
165
173
|
|
|
166
174
|
## CLI reference
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
# diffprompt
|
|
2
2
|
|
|
3
|
+
[](https://github.com/RudraDudhat2509/diffprompt/actions/workflows/ci.yml)
|
|
4
|
+
|
|
3
5
|
> git diff for your prompt's behavior
|
|
4
6
|
|
|
5
7
|
```bash
|
|
@@ -131,6 +133,12 @@ Different job. Different tool.
|
|
|
131
133
|
|
|
132
134
|
Override any layer with `--model` and `--judge`.
|
|
133
135
|
|
|
136
|
+
Groq calls are paced to stay under the free-tier rate limit. Tune it with `DIFFPROMPT_GROQ_RPM` (default 30; set `0` to disable):
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
export DIFFPROMPT_GROQ_RPM=60 # raise on paid tiers
|
|
140
|
+
```
|
|
141
|
+
|
|
134
142
|
---
|
|
135
143
|
|
|
136
144
|
## CLI reference
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
"""diffprompt — git diff for your prompt's behavior"""
|
|
2
|
-
__version__ = "0.
|
|
1
|
+
"""diffprompt — git diff for your prompt's behavior"""
|
|
2
|
+
__version__ = "0.2.0"
|
|
@@ -73,7 +73,7 @@ async def _run_diff(**kwargs):
|
|
|
73
73
|
from diffprompt.core.generator import generate_test_cases, diversity_score
|
|
74
74
|
from diffprompt.core.runner import run_both
|
|
75
75
|
from diffprompt.core.embedder import batch_similarity
|
|
76
|
-
from diffprompt.core.judge import
|
|
76
|
+
from diffprompt.core.judge import judge_all
|
|
77
77
|
from diffprompt.core.clusterer import cluster_diffs
|
|
78
78
|
from diffprompt.core.slicer import compute_slices
|
|
79
79
|
from diffprompt.core.scorer import regression_score, select_key_examples
|
|
@@ -116,26 +116,32 @@ async def _run_diff(**kwargs):
|
|
|
116
116
|
p.update(task, description=f"[green]✓[/green] {len(test_cases)} test cases diversity={div_score:.2f}")
|
|
117
117
|
|
|
118
118
|
p.update(task, description="Running both prompts...")
|
|
119
|
-
v1_results, v2_results = await run_both(
|
|
119
|
+
v1_results, v2_results = await run_both(
|
|
120
|
+
test_cases, prompt_v1, prompt_v2, model=kwargs["model"], local_only=local_only,
|
|
121
|
+
)
|
|
120
122
|
p.update(task, description=f"[green]✓[/green] {len(test_cases) * 2} completions done")
|
|
121
123
|
|
|
122
124
|
p.update(task, description="Computing semantic diff...")
|
|
123
|
-
|
|
124
|
-
|
|
125
|
+
v1_outputs = [v1_results[tc.id].output for tc in test_cases]
|
|
126
|
+
v2_outputs = [v2_results[tc.id].output for tc in test_cases]
|
|
127
|
+
similarities = batch_similarity(list(zip(v1_outputs, v2_outputs)))
|
|
128
|
+
|
|
129
|
+
if kwargs["no_judge"]:
|
|
130
|
+
judgements = [(Verdict.NEUTRAL, "judge skipped", 1.0) for _ in test_cases]
|
|
131
|
+
else:
|
|
132
|
+
judgements = await judge_all(
|
|
133
|
+
test_cases, v1_outputs, v2_outputs, similarities, local_only=local_only,
|
|
134
|
+
)
|
|
125
135
|
|
|
126
136
|
diffs = []
|
|
127
137
|
for i, tc in enumerate(test_cases):
|
|
128
|
-
|
|
129
|
-
v1_out = v1_results[tc.id].output
|
|
130
|
-
v2_out = v2_results[tc.id].output
|
|
131
|
-
if kwargs["no_judge"]:
|
|
132
|
-
verdict, reason, confidence = Verdict.NEUTRAL, "judge skipped", 1.0
|
|
133
|
-
else:
|
|
134
|
-
verdict, reason, confidence = await judge_single(tc, v1_out, v2_out, sim, local_only=local_only)
|
|
138
|
+
verdict, reason, confidence = judgements[i]
|
|
135
139
|
diffs.append(DiffResult(
|
|
136
|
-
test_case=tc, v1_output=
|
|
137
|
-
similarity=
|
|
140
|
+
test_case=tc, v1_output=v1_outputs[i], v2_output=v2_outputs[i],
|
|
141
|
+
similarity=similarities[i], divergence=1 - similarities[i],
|
|
138
142
|
verdict=verdict, reason=reason, judge_confidence=confidence,
|
|
143
|
+
v1_latency_ms=v1_results[tc.id].latency_ms,
|
|
144
|
+
v2_latency_ms=v2_results[tc.id].latency_ms,
|
|
139
145
|
))
|
|
140
146
|
p.update(task, description="[green]✓[/green] Diff complete")
|
|
141
147
|
|
|
@@ -145,6 +151,7 @@ async def _run_diff(**kwargs):
|
|
|
145
151
|
score = regression_score(diffs)
|
|
146
152
|
key_examples = await select_key_examples(
|
|
147
153
|
sorted(diffs, key=lambda d: d.divergence, reverse=True)[:20],
|
|
154
|
+
top_n=kwargs["top_n"],
|
|
148
155
|
local_only=local_only,
|
|
149
156
|
)
|
|
150
157
|
p.update(task, description="[green]✓[/green] Analysis complete")
|
|
@@ -1,130 +1,130 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Clusters diff results by embedding judge reasons.
|
|
3
|
-
Uses HDBSCAN + UMAP. Returns named failure modes.
|
|
4
|
-
"""
|
|
5
|
-
from __future__ import annotations
|
|
6
|
-
import numpy as np
|
|
7
|
-
from collections import defaultdict
|
|
8
|
-
from diffprompt.models import DiffResult, Cluster, Verdict
|
|
9
|
-
from diffprompt.core.embedder import embed
|
|
10
|
-
|
|
11
|
-
# Below this count, clustering is meaningless — return everything as unclustered
|
|
12
|
-
_MIN_CLUSTER_INPUT = 10
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
def cluster_diffs(diffs: list[DiffResult]) -> tuple[list[Cluster], list[DiffResult]]:
|
|
16
|
-
"""
|
|
17
|
-
Cluster diffs by their judge reasons.
|
|
18
|
-
Returns (clusters, unclustered) where unclustered = HDBSCAN noise (label -1).
|
|
19
|
-
"""
|
|
20
|
-
try:
|
|
21
|
-
import hdbscan
|
|
22
|
-
import umap
|
|
23
|
-
except ImportError:
|
|
24
|
-
raise ImportError("Run: pip install hdbscan umap-learn")
|
|
25
|
-
|
|
26
|
-
if len(diffs) < _MIN_CLUSTER_INPUT:
|
|
27
|
-
return [], diffs
|
|
28
|
-
|
|
29
|
-
reasons = [d.reason for d in diffs]
|
|
30
|
-
embs = embed(reasons)
|
|
31
|
-
|
|
32
|
-
# UMAP: reduce to low-dim before HDBSCAN (better cluster quality)
|
|
33
|
-
n_components = min(5, len(diffs) - 2)
|
|
34
|
-
reducer = umap.UMAP(n_components=n_components, random_state=42, verbose=False)
|
|
35
|
-
reduced = reducer.fit_transform(embs)
|
|
36
|
-
|
|
37
|
-
clusterer = hdbscan.HDBSCAN(min_cluster_size=2, min_samples=1)
|
|
38
|
-
labels = clusterer.fit_predict(reduced)
|
|
39
|
-
|
|
40
|
-
# Compute centrality per point within its cluster
|
|
41
|
-
centrality_map = _compute_centrality(embs, labels)
|
|
42
|
-
|
|
43
|
-
groups: dict[int, list[int]] = defaultdict(list)
|
|
44
|
-
for i, label in enumerate(labels):
|
|
45
|
-
groups[label].append(i)
|
|
46
|
-
|
|
47
|
-
clusters = []
|
|
48
|
-
unclustered = []
|
|
49
|
-
|
|
50
|
-
for label, indices in groups.items():
|
|
51
|
-
group_diffs = [diffs[i] for i in indices]
|
|
52
|
-
|
|
53
|
-
# Update cluster metadata on each diff
|
|
54
|
-
for i_local, i_global in enumerate(indices):
|
|
55
|
-
diffs[i_global].cluster_label = label
|
|
56
|
-
diffs[i_global].cluster_centrality = centrality_map.get(i_global, 0.0)
|
|
57
|
-
|
|
58
|
-
if label == -1:
|
|
59
|
-
unclustered.extend(group_diffs)
|
|
60
|
-
continue
|
|
61
|
-
|
|
62
|
-
cluster = Cluster(
|
|
63
|
-
label=label,
|
|
64
|
-
name=_name_cluster(label, group_diffs),
|
|
65
|
-
description=_describe_cluster(group_diffs),
|
|
66
|
-
n=len(group_diffs),
|
|
67
|
-
mean_similarity=float(np.mean([d.similarity for d in group_diffs])),
|
|
68
|
-
test_ids=[d.test_case.id for d in group_diffs],
|
|
69
|
-
)
|
|
70
|
-
clusters.append(cluster)
|
|
71
|
-
|
|
72
|
-
clusters.sort(key=lambda c: c.n, reverse=True)
|
|
73
|
-
return clusters, unclustered
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
def _compute_centrality(embs: np.ndarray, labels: np.ndarray) -> dict[int, float]:
|
|
77
|
-
"""
|
|
78
|
-
For each point, compute its mean cosine similarity to others in its cluster.
|
|
79
|
-
Returns a dict: index → centrality score (0-1).
|
|
80
|
-
"""
|
|
81
|
-
from sklearn.metrics.pairwise import cosine_similarity
|
|
82
|
-
|
|
83
|
-
centrality = {}
|
|
84
|
-
unique_labels = set(labels)
|
|
85
|
-
|
|
86
|
-
for label in unique_labels:
|
|
87
|
-
if label == -1:
|
|
88
|
-
continue
|
|
89
|
-
indices = [i for i, l in enumerate(labels) if l == label]
|
|
90
|
-
if len(indices) < 2:
|
|
91
|
-
for i in indices:
|
|
92
|
-
centrality[i] = 1.0
|
|
93
|
-
continue
|
|
94
|
-
cluster_embs = embs[indices]
|
|
95
|
-
sim_matrix = cosine_similarity(cluster_embs)
|
|
96
|
-
np.fill_diagonal(sim_matrix, 0)
|
|
97
|
-
mean_sims = sim_matrix.mean(axis=1)
|
|
98
|
-
for i_local, i_global in enumerate(indices):
|
|
99
|
-
centrality[i_global] = float(mean_sims[i_local])
|
|
100
|
-
|
|
101
|
-
return centrality
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
def _name_cluster(label: int, diffs: list[DiffResult]) -> str:
|
|
105
|
-
"""Generate a named failure mode from dominant reason keywords."""
|
|
106
|
-
reasons = " ".join(d.reason.lower() for d in diffs)
|
|
107
|
-
|
|
108
|
-
if any(w in reasons for w in ["brief", "short", "concise", "terse", "succinct"]):
|
|
109
|
-
return "BREVITY_GAIN" if _is_mostly_improvements(diffs) else "BREVITY_LOSS"
|
|
110
|
-
if any(w in reasons for w in ["context", "nuance", "detail", "omit", "missing", "incomplete"]):
|
|
111
|
-
return "CONTEXT_LOSS"
|
|
112
|
-
if any(w in reasons for w in ["refus", "declin", "won't", "cannot", "avoid"]):
|
|
113
|
-
return "REFUSAL_SHIFT"
|
|
114
|
-
if any(w in reasons for w in ["tone", "empathy", "warm", "cold", "formal", "harsh"]):
|
|
115
|
-
return "TONE_SHIFT"
|
|
116
|
-
if any(w in reasons for w in ["accura", "wrong", "incorrect", "error", "fact", "hallucin"]):
|
|
117
|
-
return "ACCURACY_CHANGE"
|
|
118
|
-
if any(w in reasons for w in ["verbos", "long", "padded", "unnecessar", "redundant"]):
|
|
119
|
-
return "VERBOSITY_GAIN"
|
|
120
|
-
if any(w in reasons for w in ["format", "structur", "bullet", "list", "markdown"]):
|
|
121
|
-
return "FORMAT_CHANGE"
|
|
122
|
-
return f"CLUSTER_{label}"
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
def _describe_cluster(diffs: list[DiffResult]) -> str:
|
|
126
|
-
return ". ".join(d.reason for d in diffs[:3])[:120]
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
def _is_mostly_improvements(diffs: list[DiffResult]) -> bool:
|
|
1
|
+
"""
|
|
2
|
+
Clusters diff results by embedding judge reasons.
|
|
3
|
+
Uses HDBSCAN + UMAP. Returns named failure modes.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import numpy as np
|
|
7
|
+
from collections import defaultdict
|
|
8
|
+
from diffprompt.models import DiffResult, Cluster, Verdict
|
|
9
|
+
from diffprompt.core.embedder import embed
|
|
10
|
+
|
|
11
|
+
# Below this count, clustering is meaningless — return everything as unclustered
|
|
12
|
+
_MIN_CLUSTER_INPUT = 10
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def cluster_diffs(diffs: list[DiffResult]) -> tuple[list[Cluster], list[DiffResult]]:
|
|
16
|
+
"""
|
|
17
|
+
Cluster diffs by their judge reasons.
|
|
18
|
+
Returns (clusters, unclustered) where unclustered = HDBSCAN noise (label -1).
|
|
19
|
+
"""
|
|
20
|
+
try:
|
|
21
|
+
import hdbscan
|
|
22
|
+
import umap
|
|
23
|
+
except ImportError:
|
|
24
|
+
raise ImportError("Run: pip install hdbscan umap-learn")
|
|
25
|
+
|
|
26
|
+
if len(diffs) < _MIN_CLUSTER_INPUT:
|
|
27
|
+
return [], diffs
|
|
28
|
+
|
|
29
|
+
reasons = [d.reason for d in diffs]
|
|
30
|
+
embs = embed(reasons)
|
|
31
|
+
|
|
32
|
+
# UMAP: reduce to low-dim before HDBSCAN (better cluster quality)
|
|
33
|
+
n_components = min(5, len(diffs) - 2)
|
|
34
|
+
reducer = umap.UMAP(n_components=n_components, random_state=42, verbose=False)
|
|
35
|
+
reduced = reducer.fit_transform(embs)
|
|
36
|
+
|
|
37
|
+
clusterer = hdbscan.HDBSCAN(min_cluster_size=2, min_samples=1)
|
|
38
|
+
labels = clusterer.fit_predict(reduced)
|
|
39
|
+
|
|
40
|
+
# Compute centrality per point within its cluster
|
|
41
|
+
centrality_map = _compute_centrality(embs, labels)
|
|
42
|
+
|
|
43
|
+
groups: dict[int, list[int]] = defaultdict(list)
|
|
44
|
+
for i, label in enumerate(labels):
|
|
45
|
+
groups[label].append(i)
|
|
46
|
+
|
|
47
|
+
clusters = []
|
|
48
|
+
unclustered = []
|
|
49
|
+
|
|
50
|
+
for label, indices in groups.items():
|
|
51
|
+
group_diffs = [diffs[i] for i in indices]
|
|
52
|
+
|
|
53
|
+
# Update cluster metadata on each diff
|
|
54
|
+
for i_local, i_global in enumerate(indices):
|
|
55
|
+
diffs[i_global].cluster_label = label
|
|
56
|
+
diffs[i_global].cluster_centrality = centrality_map.get(i_global, 0.0)
|
|
57
|
+
|
|
58
|
+
if label == -1:
|
|
59
|
+
unclustered.extend(group_diffs)
|
|
60
|
+
continue
|
|
61
|
+
|
|
62
|
+
cluster = Cluster(
|
|
63
|
+
label=label,
|
|
64
|
+
name=_name_cluster(label, group_diffs),
|
|
65
|
+
description=_describe_cluster(group_diffs),
|
|
66
|
+
n=len(group_diffs),
|
|
67
|
+
mean_similarity=float(np.mean([d.similarity for d in group_diffs])),
|
|
68
|
+
test_ids=[d.test_case.id for d in group_diffs],
|
|
69
|
+
)
|
|
70
|
+
clusters.append(cluster)
|
|
71
|
+
|
|
72
|
+
clusters.sort(key=lambda c: c.n, reverse=True)
|
|
73
|
+
return clusters, unclustered
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _compute_centrality(embs: np.ndarray, labels: np.ndarray) -> dict[int, float]:
|
|
77
|
+
"""
|
|
78
|
+
For each point, compute its mean cosine similarity to others in its cluster.
|
|
79
|
+
Returns a dict: index → centrality score (0-1).
|
|
80
|
+
"""
|
|
81
|
+
from sklearn.metrics.pairwise import cosine_similarity
|
|
82
|
+
|
|
83
|
+
centrality = {}
|
|
84
|
+
unique_labels = set(labels)
|
|
85
|
+
|
|
86
|
+
for label in unique_labels:
|
|
87
|
+
if label == -1:
|
|
88
|
+
continue
|
|
89
|
+
indices = [i for i, l in enumerate(labels) if l == label]
|
|
90
|
+
if len(indices) < 2:
|
|
91
|
+
for i in indices:
|
|
92
|
+
centrality[i] = 1.0
|
|
93
|
+
continue
|
|
94
|
+
cluster_embs = embs[indices]
|
|
95
|
+
sim_matrix = cosine_similarity(cluster_embs)
|
|
96
|
+
np.fill_diagonal(sim_matrix, 0)
|
|
97
|
+
mean_sims = sim_matrix.mean(axis=1)
|
|
98
|
+
for i_local, i_global in enumerate(indices):
|
|
99
|
+
centrality[i_global] = float(mean_sims[i_local])
|
|
100
|
+
|
|
101
|
+
return centrality
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _name_cluster(label: int, diffs: list[DiffResult]) -> str:
|
|
105
|
+
"""Generate a named failure mode from dominant reason keywords."""
|
|
106
|
+
reasons = " ".join(d.reason.lower() for d in diffs)
|
|
107
|
+
|
|
108
|
+
if any(w in reasons for w in ["brief", "short", "concise", "terse", "succinct"]):
|
|
109
|
+
return "BREVITY_GAIN" if _is_mostly_improvements(diffs) else "BREVITY_LOSS"
|
|
110
|
+
if any(w in reasons for w in ["context", "nuance", "detail", "omit", "missing", "incomplete"]):
|
|
111
|
+
return "CONTEXT_LOSS"
|
|
112
|
+
if any(w in reasons for w in ["refus", "declin", "won't", "cannot", "avoid"]):
|
|
113
|
+
return "REFUSAL_SHIFT"
|
|
114
|
+
if any(w in reasons for w in ["tone", "empathy", "warm", "cold", "formal", "harsh"]):
|
|
115
|
+
return "TONE_SHIFT"
|
|
116
|
+
if any(w in reasons for w in ["accura", "wrong", "incorrect", "error", "fact", "hallucin"]):
|
|
117
|
+
return "ACCURACY_CHANGE"
|
|
118
|
+
if any(w in reasons for w in ["verbos", "long", "padded", "unnecessar", "redundant"]):
|
|
119
|
+
return "VERBOSITY_GAIN"
|
|
120
|
+
if any(w in reasons for w in ["format", "structur", "bullet", "list", "markdown"]):
|
|
121
|
+
return "FORMAT_CHANGE"
|
|
122
|
+
return f"CLUSTER_{label}"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _describe_cluster(diffs: list[DiffResult]) -> str:
|
|
126
|
+
return ". ".join(d.reason for d in diffs[:3])[:120]
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _is_mostly_improvements(diffs: list[DiffResult]) -> bool:
|
|
130
130
|
return sum(1 for d in diffs if d.verdict == Verdict.IMPROVEMENT) > len(diffs) / 2
|
|
@@ -1,63 +1,63 @@
|
|
|
1
|
-
"""
|
|
2
|
-
Embedding + similarity layer.
|
|
3
|
-
All local, all free. No API calls.
|
|
4
|
-
"""
|
|
5
|
-
from __future__ import annotations
|
|
6
|
-
import logging
|
|
7
|
-
import os
|
|
8
|
-
import warnings
|
|
9
|
-
import numpy as np
|
|
10
|
-
from functools import lru_cache
|
|
11
|
-
|
|
12
|
-
# Suppress noisy warnings from HuggingFace / sentence-transformers / UMAP
|
|
13
|
-
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
|
|
14
|
-
os.environ.setdefault("HF_HUB_DISABLE_SYMLINKS_WARNING", "1")
|
|
15
|
-
logging.getLogger("sentence_transformers").setLevel(logging.ERROR)
|
|
16
|
-
logging.getLogger("transformers").setLevel(logging.ERROR)
|
|
17
|
-
logging.getLogger("huggingface_hub").setLevel(logging.ERROR)
|
|
18
|
-
warnings.filterwarnings("ignore", message=".*n_jobs value.*overridden.*")
|
|
19
|
-
warnings.filterwarnings("ignore", message=".*unauthenticated.*")
|
|
20
|
-
warnings.filterwarnings("ignore", message=".*UNEXPECTED.*")
|
|
21
|
-
warnings.filterwarnings("ignore", category=UserWarning, module="umap")
|
|
22
|
-
|
|
23
|
-
from sentence_transformers import SentenceTransformer
|
|
24
|
-
from sklearn.metrics.pairwise import cosine_similarity as sk_cosine
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
MODEL_NAME = "all-MiniLM-L6-v2"
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
@lru_cache(maxsize=1)
|
|
31
|
-
def get_embedder() -> SentenceTransformer:
|
|
32
|
-
"""Lazy-load embedder. Cached so it only loads once per session."""
|
|
33
|
-
# Silence the load report printed to stdout by newer sentence-transformers
|
|
34
|
-
import io, contextlib
|
|
35
|
-
f = io.StringIO()
|
|
36
|
-
with contextlib.redirect_stdout(f), contextlib.redirect_stderr(f):
|
|
37
|
-
model = SentenceTransformer(MODEL_NAME)
|
|
38
|
-
return model
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
def embed(texts: list[str]) -> np.ndarray:
|
|
42
|
-
return get_embedder().encode(texts, show_progress_bar=False)
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
def similarity(text_a: str, text_b: str) -> float:
|
|
46
|
-
"""Cosine similarity between two texts. Returns float 0-1."""
|
|
47
|
-
embs = embed([text_a, text_b])
|
|
48
|
-
score = sk_cosine([embs[0]], [embs[1]])[0][0]
|
|
49
|
-
return float(np.clip(score, 0, 1))
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
def batch_similarity(pairs: list[tuple[str, str]]) -> list[float]:
|
|
53
|
-
"""
|
|
54
|
-
Efficient batch similarity for many pairs.
|
|
55
|
-
Embeds all texts in one pass instead of N passes.
|
|
56
|
-
"""
|
|
57
|
-
all_texts = [t for pair in pairs for t in pair]
|
|
58
|
-
all_embs = embed(all_texts)
|
|
59
|
-
scores = []
|
|
60
|
-
for i in range(0, len(all_embs), 2):
|
|
61
|
-
score = sk_cosine([all_embs[i]], [all_embs[i + 1]])[0][0]
|
|
62
|
-
scores.append(float(np.clip(score, 0, 1)))
|
|
1
|
+
"""
|
|
2
|
+
Embedding + similarity layer.
|
|
3
|
+
All local, all free. No API calls.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import warnings
|
|
9
|
+
import numpy as np
|
|
10
|
+
from functools import lru_cache
|
|
11
|
+
|
|
12
|
+
# Suppress noisy warnings from HuggingFace / sentence-transformers / UMAP
|
|
13
|
+
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
|
|
14
|
+
os.environ.setdefault("HF_HUB_DISABLE_SYMLINKS_WARNING", "1")
|
|
15
|
+
logging.getLogger("sentence_transformers").setLevel(logging.ERROR)
|
|
16
|
+
logging.getLogger("transformers").setLevel(logging.ERROR)
|
|
17
|
+
logging.getLogger("huggingface_hub").setLevel(logging.ERROR)
|
|
18
|
+
warnings.filterwarnings("ignore", message=".*n_jobs value.*overridden.*")
|
|
19
|
+
warnings.filterwarnings("ignore", message=".*unauthenticated.*")
|
|
20
|
+
warnings.filterwarnings("ignore", message=".*UNEXPECTED.*")
|
|
21
|
+
warnings.filterwarnings("ignore", category=UserWarning, module="umap")
|
|
22
|
+
|
|
23
|
+
from sentence_transformers import SentenceTransformer
|
|
24
|
+
from sklearn.metrics.pairwise import cosine_similarity as sk_cosine
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
MODEL_NAME = "all-MiniLM-L6-v2"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@lru_cache(maxsize=1)
|
|
31
|
+
def get_embedder() -> SentenceTransformer:
|
|
32
|
+
"""Lazy-load embedder. Cached so it only loads once per session."""
|
|
33
|
+
# Silence the load report printed to stdout by newer sentence-transformers
|
|
34
|
+
import io, contextlib
|
|
35
|
+
f = io.StringIO()
|
|
36
|
+
with contextlib.redirect_stdout(f), contextlib.redirect_stderr(f):
|
|
37
|
+
model = SentenceTransformer(MODEL_NAME)
|
|
38
|
+
return model
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def embed(texts: list[str]) -> np.ndarray:
|
|
42
|
+
return get_embedder().encode(texts, show_progress_bar=False)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def similarity(text_a: str, text_b: str) -> float:
|
|
46
|
+
"""Cosine similarity between two texts. Returns float 0-1."""
|
|
47
|
+
embs = embed([text_a, text_b])
|
|
48
|
+
score = sk_cosine([embs[0]], [embs[1]])[0][0]
|
|
49
|
+
return float(np.clip(score, 0, 1))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def batch_similarity(pairs: list[tuple[str, str]]) -> list[float]:
|
|
53
|
+
"""
|
|
54
|
+
Efficient batch similarity for many pairs.
|
|
55
|
+
Embeds all texts in one pass instead of N passes.
|
|
56
|
+
"""
|
|
57
|
+
all_texts = [t for pair in pairs for t in pair]
|
|
58
|
+
all_embs = embed(all_texts)
|
|
59
|
+
scores = []
|
|
60
|
+
for i in range(0, len(all_embs), 2):
|
|
61
|
+
score = sk_cosine([all_embs[i]], [all_embs[i + 1]])[0][0]
|
|
62
|
+
scores.append(float(np.clip(score, 0, 1)))
|
|
63
63
|
return scores
|