diffprompt 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffprompt/__init__.py +2 -0
- diffprompt/cli.py +246 -0
- diffprompt/core/__init__.py +0 -0
- diffprompt/core/clusterer.py +130 -0
- diffprompt/core/embedder.py +63 -0
- diffprompt/core/generator.py +123 -0
- diffprompt/core/judge.py +94 -0
- diffprompt/core/ontology.py +160 -0
- diffprompt/core/runner.py +65 -0
- diffprompt/core/scorer.py +102 -0
- diffprompt/core/slicer.py +152 -0
- diffprompt/models/__init__.py +105 -0
- diffprompt/models/cascade.py +144 -0
- diffprompt/output/__init__.py +0 -0
- diffprompt/output/exporter.py +227 -0
- diffprompt/output/terminal.py +154 -0
- diffprompt-0.1.0.dist-info/METADATA +249 -0
- diffprompt-0.1.0.dist-info/RECORD +20 -0
- diffprompt-0.1.0.dist-info/WHEEL +4 -0
- diffprompt-0.1.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Ontology inference and anchor-based tagging.
|
|
3
|
+
|
|
4
|
+
Two steps:
|
|
5
|
+
1. Infer dimensions + tags from the prompt (one LLM call)
|
|
6
|
+
2. Generate anchor sentences per tag (one LLM call per tag)
|
|
7
|
+
3. Tag test inputs using embedding similarity to anchors (no LLM)
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
import json
|
|
11
|
+
import re
|
|
12
|
+
from typing import Optional
|
|
13
|
+
from sentence_transformers import SentenceTransformer
|
|
14
|
+
from sklearn.metrics.pairwise import cosine_similarity
|
|
15
|
+
import numpy as np
|
|
16
|
+
|
|
17
|
+
from diffprompt.models.cascade import call_cascade
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
INFER_PROMPT = """You are designing a test suite for an LLM prompt.
|
|
21
|
+
|
|
22
|
+
Prompt: {prompt}
|
|
23
|
+
|
|
24
|
+
Identify 3-4 dimensions that would reveal BEHAVIORAL differences in how this prompt responds.
|
|
25
|
+
Focus on: user intent, emotional state, topic complexity, and request type.
|
|
26
|
+
Avoid surface-level dimensions like length, format, or punctuation style.
|
|
27
|
+
|
|
28
|
+
Return ONLY valid JSON. Example:
|
|
29
|
+
{{
|
|
30
|
+
"tone": ["formal", "casual", "emotional", "urgent"],
|
|
31
|
+
"complexity": ["simple", "multi-part", "ambiguous"],
|
|
32
|
+
"intent": ["lookup", "reasoning", "emotional-support"]
|
|
33
|
+
}}
|
|
34
|
+
|
|
35
|
+
Each dimension must predict a meaningfully different response from this specific prompt."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
ANCHOR_PROMPT = """For the tag "{tag}" in dimension "{dimension}", write 3 short example sentences
|
|
39
|
+
that real users would send to this prompt: {prompt}
|
|
40
|
+
|
|
41
|
+
Each sentence must clearly represent "{tag}" and be distinct from each other.
|
|
42
|
+
Return ONLY a JSON array of 3 strings. No explanation."""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _extract_json_object(raw: str) -> str:
|
|
46
|
+
"""Extract the first {...} block from raw LLM output."""
|
|
47
|
+
clean = re.sub(r"```(?:json)?|```", "", raw).strip()
|
|
48
|
+
match = re.search(r'\{.*\}', clean, re.DOTALL)
|
|
49
|
+
return match.group(0) if match else clean
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _extract_json_array(raw: str) -> str:
|
|
53
|
+
"""Extract the first [...] block from raw LLM output."""
|
|
54
|
+
clean = re.sub(r"```(?:json)?|```", "", raw).strip()
|
|
55
|
+
match = re.search(r'\[.*\]', clean, re.DOTALL)
|
|
56
|
+
return match.group(0) if match else clean
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class Ontology:
|
|
60
|
+
def __init__(self):
|
|
61
|
+
self.dimensions: dict[str, list[str]] = {}
|
|
62
|
+
self.anchors: dict[str, dict[str, list[str]]] = {}
|
|
63
|
+
self.anchor_embeddings: dict[str, dict[str, np.ndarray]] = {}
|
|
64
|
+
self._embedder: Optional[SentenceTransformer] = None
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def embedder(self) -> SentenceTransformer:
|
|
68
|
+
if self._embedder is None:
|
|
69
|
+
import io, contextlib
|
|
70
|
+
f = io.StringIO()
|
|
71
|
+
with contextlib.redirect_stdout(f), contextlib.redirect_stderr(f):
|
|
72
|
+
self._embedder = SentenceTransformer("all-MiniLM-L6-v2")
|
|
73
|
+
return self._embedder
|
|
74
|
+
|
|
75
|
+
async def infer(self, prompt: str, local_only: bool = False) -> None:
|
|
76
|
+
"""Infer dimensions and tags from the prompt. One LLM call."""
|
|
77
|
+
raw, _ = await call_cascade(
|
|
78
|
+
INFER_PROMPT.format(prompt=prompt),
|
|
79
|
+
local_only=local_only,
|
|
80
|
+
)
|
|
81
|
+
try:
|
|
82
|
+
parsed = json.loads(_extract_json_object(raw))
|
|
83
|
+
if len(parsed) == 1 and isinstance(list(parsed.values())[0], dict):
|
|
84
|
+
parsed = list(parsed.values())[0]
|
|
85
|
+
self.dimensions = {k: v for k, v in parsed.items() if isinstance(v, list)}
|
|
86
|
+
except json.JSONDecodeError:
|
|
87
|
+
self.dimensions = {
|
|
88
|
+
"user_intent": ["informational", "transactional", "complaint", "other"],
|
|
89
|
+
"emotional_state": ["neutral", "frustrated", "confused", "urgent"],
|
|
90
|
+
"request_type": ["specific", "open_ended", "troubleshooting"],
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
async def build_anchors(self, prompt: str, local_only: bool = False) -> None:
|
|
94
|
+
"""Generate 3 anchor sentences per tag. One LLM call per tag."""
|
|
95
|
+
for dimension, tags in self.dimensions.items():
|
|
96
|
+
self.anchors[dimension] = {}
|
|
97
|
+
for tag in tags:
|
|
98
|
+
raw, _ = await call_cascade(
|
|
99
|
+
ANCHOR_PROMPT.format(tag=tag, dimension=dimension, prompt=prompt),
|
|
100
|
+
local_only=local_only,
|
|
101
|
+
)
|
|
102
|
+
try:
|
|
103
|
+
parsed = json.loads(_extract_json_array(raw))
|
|
104
|
+
if isinstance(parsed, list) and parsed:
|
|
105
|
+
self.anchors[dimension][tag] = [str(s) for s in parsed]
|
|
106
|
+
else:
|
|
107
|
+
self.anchors[dimension][tag] = [raw.strip()[:200]]
|
|
108
|
+
except Exception:
|
|
109
|
+
self.anchors[dimension][tag] = [raw.strip()[:200]]
|
|
110
|
+
|
|
111
|
+
# Build embeddings for all anchors
|
|
112
|
+
for dimension, tag_anchors in self.anchors.items():
|
|
113
|
+
self.anchor_embeddings[dimension] = {}
|
|
114
|
+
for tag, sentences in tag_anchors.items():
|
|
115
|
+
self.anchor_embeddings[dimension][tag] = self.embedder.encode(sentences)
|
|
116
|
+
|
|
117
|
+
def tag(self, input_text: str) -> dict[str, str]:
|
|
118
|
+
"""
|
|
119
|
+
Tag an input using max similarity across multiple anchors per tag.
|
|
120
|
+
No LLM call — pure embeddings. Deterministic and fast.
|
|
121
|
+
"""
|
|
122
|
+
if not self.anchor_embeddings:
|
|
123
|
+
return {}
|
|
124
|
+
|
|
125
|
+
input_emb = self.embedder.encode([input_text])
|
|
126
|
+
result = {}
|
|
127
|
+
|
|
128
|
+
for dimension, tags in self.dimensions.items():
|
|
129
|
+
if dimension not in self.anchor_embeddings:
|
|
130
|
+
continue
|
|
131
|
+
best_tag = tags[0]
|
|
132
|
+
best_score = -1.0
|
|
133
|
+
|
|
134
|
+
for tag in tags:
|
|
135
|
+
if tag not in self.anchor_embeddings[dimension]:
|
|
136
|
+
continue
|
|
137
|
+
anchor_embs = self.anchor_embeddings[dimension][tag]
|
|
138
|
+
sims = cosine_similarity(input_emb, anchor_embs)[0]
|
|
139
|
+
score = float(np.max(sims))
|
|
140
|
+
if score > best_score:
|
|
141
|
+
best_score = score
|
|
142
|
+
best_tag = tag
|
|
143
|
+
|
|
144
|
+
result[dimension] = best_tag
|
|
145
|
+
|
|
146
|
+
return result
|
|
147
|
+
|
|
148
|
+
def to_dict(self) -> dict:
|
|
149
|
+
return {"dimensions": self.dimensions, "anchors": self.anchors}
|
|
150
|
+
|
|
151
|
+
@classmethod
|
|
152
|
+
def from_dict(cls, data: dict) -> "Ontology":
|
|
153
|
+
o = cls()
|
|
154
|
+
o.dimensions = data["dimensions"]
|
|
155
|
+
o.anchors = data["anchors"]
|
|
156
|
+
for dimension, tag_anchors in o.anchors.items():
|
|
157
|
+
o.anchor_embeddings[dimension] = {}
|
|
158
|
+
for tag, sentences in tag_anchors.items():
|
|
159
|
+
o.anchor_embeddings[dimension][tag] = o.embedder.encode(sentences)
|
|
160
|
+
return o
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Runs both prompts on all test cases.
|
|
3
|
+
Returns RunResult for each (test_case, prompt_version) pair.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
import asyncio
|
|
7
|
+
import time
|
|
8
|
+
from diffprompt.models import TestCase, RunResult
|
|
9
|
+
from diffprompt.models.cascade import call_cascade
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
async def run_single(
|
|
13
|
+
test_case: TestCase,
|
|
14
|
+
prompt: str,
|
|
15
|
+
version: str,
|
|
16
|
+
model: str,
|
|
17
|
+
local_only: bool = False,
|
|
18
|
+
) -> RunResult:
|
|
19
|
+
start = time.monotonic()
|
|
20
|
+
output, model_used = await call_cascade(
|
|
21
|
+
test_case.input,
|
|
22
|
+
system=prompt,
|
|
23
|
+
local_only=local_only,
|
|
24
|
+
)
|
|
25
|
+
latency_ms = (time.monotonic() - start) * 1000
|
|
26
|
+
|
|
27
|
+
return RunResult(
|
|
28
|
+
test_id=test_case.id,
|
|
29
|
+
prompt_version=version,
|
|
30
|
+
output=output,
|
|
31
|
+
model_used=model_used,
|
|
32
|
+
latency_ms=latency_ms,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
async def run_both(
|
|
37
|
+
test_cases: list[TestCase],
|
|
38
|
+
prompt_v1: str,
|
|
39
|
+
prompt_v2: str,
|
|
40
|
+
model: str = "groq/llama-3.3-70b-versatile",
|
|
41
|
+
local_only: bool = False,
|
|
42
|
+
concurrency: int = 5,
|
|
43
|
+
) -> tuple[dict[str, RunResult], dict[str, RunResult]]:
|
|
44
|
+
"""
|
|
45
|
+
Run both prompts on all test cases concurrently.
|
|
46
|
+
Returns (v1_results, v2_results) as dicts keyed by test_id.
|
|
47
|
+
"""
|
|
48
|
+
semaphore = asyncio.Semaphore(concurrency)
|
|
49
|
+
|
|
50
|
+
async def run_with_sem(tc, prompt, version):
|
|
51
|
+
async with semaphore:
|
|
52
|
+
return await run_single(tc, prompt, version, model, local_only)
|
|
53
|
+
|
|
54
|
+
# Fire all tasks concurrently (bounded by semaphore)
|
|
55
|
+
v1_tasks = [run_with_sem(tc, prompt_v1, "v1") for tc in test_cases]
|
|
56
|
+
v2_tasks = [run_with_sem(tc, prompt_v2, "v2") for tc in test_cases]
|
|
57
|
+
|
|
58
|
+
v1_raw, v2_raw = await asyncio.gather(
|
|
59
|
+
asyncio.gather(*v1_tasks),
|
|
60
|
+
asyncio.gather(*v2_tasks),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
v1_results = {r.test_id: r for r in v1_raw}
|
|
64
|
+
v2_results = {r.test_id: r for r in v2_raw}
|
|
65
|
+
return v1_results, v2_results
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Regression scoring and key example selection.
|
|
3
|
+
"""
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
import numpy as np
|
|
6
|
+
from diffprompt.models import DiffResult, KeyExample, Verdict
|
|
7
|
+
from diffprompt.models.cascade import call_cascade
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
WHY_IT_MATTERS_PROMPT = """
|
|
11
|
+
Input: {input}
|
|
12
|
+
V1 output: {v1}
|
|
13
|
+
V2 output: {v2}
|
|
14
|
+
Verdict: {verdict}
|
|
15
|
+
|
|
16
|
+
In ONE sentence, explain what this reveals about the prompt change.
|
|
17
|
+
Focus on the mechanism — why did v2 behave differently here?
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def regression_score(diffs: list[DiffResult]) -> float:
|
|
22
|
+
if not diffs:
|
|
23
|
+
return 50.0
|
|
24
|
+
total_weight = sum(d.divergence for d in diffs)
|
|
25
|
+
if total_weight == 0:
|
|
26
|
+
return 100.0
|
|
27
|
+
weighted_sum = sum(
|
|
28
|
+
d.divergence * (1.0 if d.verdict == Verdict.IMPROVEMENT else -1.0 if d.verdict == Verdict.REGRESSION else 0.0)
|
|
29
|
+
for d in diffs
|
|
30
|
+
)
|
|
31
|
+
return float(round(((weighted_sum / total_weight) + 1) / 2 * 100, 1))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def importance_score(diff: DiffResult) -> float:
|
|
35
|
+
divergence = diff.divergence
|
|
36
|
+
centrality = diff.cluster_centrality
|
|
37
|
+
input_length = len(diff.test_case.input.split())
|
|
38
|
+
simplicity = max(0.0, 1.0 - input_length / 50)
|
|
39
|
+
surprise = divergence * simplicity
|
|
40
|
+
return 0.4 * divergence + 0.3 * centrality + 0.3 * surprise
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
async def select_key_examples(
|
|
44
|
+
diffs: list[DiffResult],
|
|
45
|
+
top_n: int = 3,
|
|
46
|
+
local_only: bool = False,
|
|
47
|
+
) -> list[KeyExample]:
|
|
48
|
+
if not diffs:
|
|
49
|
+
return []
|
|
50
|
+
|
|
51
|
+
for d in diffs:
|
|
52
|
+
d.importance_score = importance_score(d)
|
|
53
|
+
|
|
54
|
+
selected: list[tuple[str, DiffResult]] = []
|
|
55
|
+
used_ids: set[str] = set()
|
|
56
|
+
|
|
57
|
+
def _pick(slot: str, candidate) -> None:
|
|
58
|
+
if candidate and candidate.test_case.id not in used_ids:
|
|
59
|
+
selected.append((slot, candidate))
|
|
60
|
+
used_ids.add(candidate.test_case.id)
|
|
61
|
+
|
|
62
|
+
_pick("most_important", max(diffs, key=lambda d: d.importance_score))
|
|
63
|
+
|
|
64
|
+
improvements = [d for d in diffs if d.verdict == Verdict.IMPROVEMENT and d.test_case.id not in used_ids]
|
|
65
|
+
if improvements:
|
|
66
|
+
_pick("best_improvement", max(improvements, key=lambda d: d.divergence))
|
|
67
|
+
|
|
68
|
+
remaining = [d for d in diffs if d.test_case.id not in used_ids]
|
|
69
|
+
if remaining:
|
|
70
|
+
_pick("most_surprising", max(
|
|
71
|
+
remaining,
|
|
72
|
+
key=lambda d: d.divergence * max(0, 1 - len(d.test_case.input.split()) / 50),
|
|
73
|
+
))
|
|
74
|
+
|
|
75
|
+
if top_n > 3:
|
|
76
|
+
regressions = sorted(
|
|
77
|
+
[d for d in diffs if d.verdict == Verdict.REGRESSION and d.test_case.id not in used_ids],
|
|
78
|
+
key=lambda d: d.divergence, reverse=True,
|
|
79
|
+
)
|
|
80
|
+
for d in regressions[:top_n - 3]:
|
|
81
|
+
_pick(f"regression_{len(selected)}", d)
|
|
82
|
+
|
|
83
|
+
selected = selected[:top_n]
|
|
84
|
+
|
|
85
|
+
examples = []
|
|
86
|
+
for slot_name, diff in selected:
|
|
87
|
+
why, _ = await call_cascade(
|
|
88
|
+
WHY_IT_MATTERS_PROMPT.format(
|
|
89
|
+
input=diff.test_case.input,
|
|
90
|
+
v1=diff.v1_output[:600],
|
|
91
|
+
v2=diff.v2_output[:600],
|
|
92
|
+
verdict=diff.verdict.value,
|
|
93
|
+
),
|
|
94
|
+
local_only=local_only,
|
|
95
|
+
)
|
|
96
|
+
examples.append(KeyExample(
|
|
97
|
+
slot=slot_name,
|
|
98
|
+
diff=diff,
|
|
99
|
+
why_it_matters=why.strip(),
|
|
100
|
+
))
|
|
101
|
+
|
|
102
|
+
return examples
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Behavioral slicing.
|
|
3
|
+
Groups diffs by input tags and computes per-slice performance.
|
|
4
|
+
Recursive splitting for high-variance slices (max depth 3).
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
import numpy as np
|
|
8
|
+
from diffprompt.models import DiffResult, SliceResult, Verdict, TestCategory
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
MIN_SLICE_SIZE = 5
|
|
12
|
+
MAX_DEPTH = 3
|
|
13
|
+
VARIANCE_THRESHOLD = 0.1 # stop splitting if variance is already low
|
|
14
|
+
HIGH_VARIANCE_THRESHOLD = 0.15 # trigger recursive split above this
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def compute_slices(diffs: list[DiffResult]) -> list[SliceResult]:
|
|
18
|
+
"""Compute behavioral slices across all tag dimensions."""
|
|
19
|
+
if not diffs or not diffs[0].test_case.tags:
|
|
20
|
+
return []
|
|
21
|
+
|
|
22
|
+
dimensions = list(diffs[0].test_case.tags.keys())
|
|
23
|
+
slices = []
|
|
24
|
+
|
|
25
|
+
for dimension in dimensions:
|
|
26
|
+
# Group by tag value
|
|
27
|
+
groups: dict[str, list[DiffResult]] = {}
|
|
28
|
+
for d in diffs:
|
|
29
|
+
val = d.test_case.tags.get(dimension, "unknown")
|
|
30
|
+
groups.setdefault(val, []).append(d)
|
|
31
|
+
|
|
32
|
+
for value, group_diffs in groups.items():
|
|
33
|
+
slice_result = _compute_slice(
|
|
34
|
+
dimension=dimension,
|
|
35
|
+
value=value,
|
|
36
|
+
diffs=group_diffs,
|
|
37
|
+
depth=1,
|
|
38
|
+
)
|
|
39
|
+
if slice_result:
|
|
40
|
+
slices.append(slice_result)
|
|
41
|
+
|
|
42
|
+
# Recursive split if high variance + enough data + not too deep
|
|
43
|
+
if (
|
|
44
|
+
slice_result
|
|
45
|
+
and slice_result.variance > HIGH_VARIANCE_THRESHOLD
|
|
46
|
+
and len(group_diffs) >= MIN_SLICE_SIZE * 2
|
|
47
|
+
and slice_result.depth < MAX_DEPTH
|
|
48
|
+
):
|
|
49
|
+
sub_slices = _recursive_split(group_diffs, dimension, depth=2)
|
|
50
|
+
slices.extend(sub_slices)
|
|
51
|
+
|
|
52
|
+
# Sort by mean_similarity ascending (worst slices first)
|
|
53
|
+
slices.sort(key=lambda s: s.mean_similarity)
|
|
54
|
+
return slices
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _compute_slice(
|
|
58
|
+
dimension: str,
|
|
59
|
+
value: str,
|
|
60
|
+
diffs: list[DiffResult],
|
|
61
|
+
depth: int,
|
|
62
|
+
) -> SliceResult | None:
|
|
63
|
+
if len(diffs) < 2:
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
sims = [d.similarity for d in diffs]
|
|
67
|
+
mean_sim = float(np.mean(sims))
|
|
68
|
+
variance = float(np.var(sims))
|
|
69
|
+
|
|
70
|
+
typical_ratio = sum(
|
|
71
|
+
1 for d in diffs if d.test_case.category == TestCategory.TYPICAL
|
|
72
|
+
) / len(diffs)
|
|
73
|
+
|
|
74
|
+
confidence = _compute_confidence(
|
|
75
|
+
n=len(diffs),
|
|
76
|
+
variance=variance,
|
|
77
|
+
typical_ratio=typical_ratio,
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
verdicts = [d.verdict for d in diffs]
|
|
81
|
+
verdict = _aggregate_verdict(verdicts)
|
|
82
|
+
|
|
83
|
+
return SliceResult(
|
|
84
|
+
dimension=dimension,
|
|
85
|
+
value=value,
|
|
86
|
+
label=f"{dimension}:{value}",
|
|
87
|
+
n=len(diffs),
|
|
88
|
+
mean_similarity=mean_sim,
|
|
89
|
+
variance=variance,
|
|
90
|
+
typical_ratio=typical_ratio,
|
|
91
|
+
confidence=confidence,
|
|
92
|
+
verdict=verdict,
|
|
93
|
+
depth=depth,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _recursive_split(
|
|
98
|
+
diffs: list[DiffResult],
|
|
99
|
+
parent_dimension: str,
|
|
100
|
+
depth: int,
|
|
101
|
+
) -> list[SliceResult]:
|
|
102
|
+
"""Try splitting a high-variance slice by other dimensions."""
|
|
103
|
+
if depth > MAX_DEPTH or not diffs:
|
|
104
|
+
return []
|
|
105
|
+
|
|
106
|
+
other_dimensions = [
|
|
107
|
+
k for k in diffs[0].test_case.tags.keys()
|
|
108
|
+
if k != parent_dimension
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
results = []
|
|
112
|
+
for dimension in other_dimensions:
|
|
113
|
+
groups: dict[str, list[DiffResult]] = {}
|
|
114
|
+
for d in diffs:
|
|
115
|
+
val = d.test_case.tags.get(dimension, "unknown")
|
|
116
|
+
groups.setdefault(val, []).append(d)
|
|
117
|
+
|
|
118
|
+
for value, group in groups.items():
|
|
119
|
+
if len(group) < MIN_SLICE_SIZE:
|
|
120
|
+
continue
|
|
121
|
+
slice_result = _compute_slice(
|
|
122
|
+
dimension=f"{parent_dimension}+{dimension}",
|
|
123
|
+
value=value,
|
|
124
|
+
diffs=group,
|
|
125
|
+
depth=depth,
|
|
126
|
+
)
|
|
127
|
+
if slice_result:
|
|
128
|
+
results.append(slice_result)
|
|
129
|
+
|
|
130
|
+
return results
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _compute_confidence(n: int, variance: float, typical_ratio: float) -> float:
|
|
134
|
+
"""
|
|
135
|
+
Confidence = f(variance, typical_ratio, n).
|
|
136
|
+
Variance is checked first — high variance = impure slice = low confidence.
|
|
137
|
+
"""
|
|
138
|
+
# Variance penalty (primary signal)
|
|
139
|
+
variance_score = max(0.0, 1.0 - variance * 5)
|
|
140
|
+
|
|
141
|
+
# Typical ratio bonus (real-world signal)
|
|
142
|
+
typical_score = typical_ratio
|
|
143
|
+
|
|
144
|
+
# N score (statistical power)
|
|
145
|
+
n_score = min(1.0, n / 20)
|
|
146
|
+
|
|
147
|
+
return float(0.5 * variance_score + 0.3 * typical_score + 0.2 * n_score)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _aggregate_verdict(verdicts: list[Verdict]) -> Verdict:
|
|
151
|
+
counts = {v: verdicts.count(v) for v in Verdict}
|
|
152
|
+
return max(counts, key=counts.get)
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Core data models for diffprompt.
|
|
3
|
+
These types flow through the entire pipeline — generator → runner → diff → analysis → output.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
from enum import Enum
|
|
7
|
+
from typing import Optional
|
|
8
|
+
from pydantic import BaseModel, Field
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class TestCategory(str, Enum):
|
|
12
|
+
TYPICAL = "typical"
|
|
13
|
+
BOUNDARY = "boundary"
|
|
14
|
+
ADVERSARIAL = "adversarial"
|
|
15
|
+
FORMAT = "format"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class Verdict(str, Enum):
|
|
19
|
+
IMPROVEMENT = "improvement"
|
|
20
|
+
REGRESSION = "regression"
|
|
21
|
+
NEUTRAL = "neutral"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class OutputFormat(str, Enum):
|
|
25
|
+
TERMINAL = "terminal"
|
|
26
|
+
JSON = "json"
|
|
27
|
+
HTML = "html"
|
|
28
|
+
TXT = "txt"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class TestCase(BaseModel):
|
|
32
|
+
id: str
|
|
33
|
+
input: str
|
|
34
|
+
category: TestCategory
|
|
35
|
+
tags: dict[str, str] = Field(default_factory=dict)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class RunResult(BaseModel):
|
|
39
|
+
test_id: str
|
|
40
|
+
prompt_version: str
|
|
41
|
+
output: str
|
|
42
|
+
model_used: str
|
|
43
|
+
latency_ms: Optional[float] = None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class DiffResult(BaseModel):
|
|
47
|
+
test_case: TestCase
|
|
48
|
+
v1_output: str
|
|
49
|
+
v2_output: str
|
|
50
|
+
similarity: float
|
|
51
|
+
divergence: float
|
|
52
|
+
verdict: Verdict
|
|
53
|
+
reason: str
|
|
54
|
+
judge_confidence: float
|
|
55
|
+
importance_score: float = 0.0
|
|
56
|
+
cluster_label: int = -1
|
|
57
|
+
cluster_centrality: float = 0.0
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class SliceResult(BaseModel):
|
|
61
|
+
dimension: str
|
|
62
|
+
value: str
|
|
63
|
+
label: str
|
|
64
|
+
n: int
|
|
65
|
+
mean_similarity: float
|
|
66
|
+
variance: float
|
|
67
|
+
typical_ratio: float
|
|
68
|
+
confidence: float
|
|
69
|
+
verdict: Verdict
|
|
70
|
+
depth: int = 1
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
class Cluster(BaseModel):
|
|
74
|
+
label: int
|
|
75
|
+
name: str
|
|
76
|
+
description: str
|
|
77
|
+
n: int
|
|
78
|
+
mean_similarity: float
|
|
79
|
+
test_ids: list[str]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class KeyExample(BaseModel):
|
|
83
|
+
slot: str
|
|
84
|
+
diff: DiffResult
|
|
85
|
+
why_it_matters: str
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class DiffReport(BaseModel):
|
|
89
|
+
prompt_v1: str
|
|
90
|
+
prompt_v2: str
|
|
91
|
+
model: str
|
|
92
|
+
judge: str
|
|
93
|
+
test_cases: list[TestCase]
|
|
94
|
+
diversity_score: float
|
|
95
|
+
diffs: list[DiffResult]
|
|
96
|
+
slices: list[SliceResult]
|
|
97
|
+
clusters: list[Cluster]
|
|
98
|
+
unclustered: list[DiffResult]
|
|
99
|
+
key_examples: list[KeyExample]
|
|
100
|
+
regression_score: float
|
|
101
|
+
n_improved: int
|
|
102
|
+
n_regressed: int
|
|
103
|
+
n_neutral: int
|
|
104
|
+
verdict: Verdict
|
|
105
|
+
recommendation: str
|