diffprompt 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,160 @@
1
+ """
2
+ Ontology inference and anchor-based tagging.
3
+
4
+ Two steps:
5
+ 1. Infer dimensions + tags from the prompt (one LLM call)
6
+ 2. Generate anchor sentences per tag (one LLM call per tag)
7
+ 3. Tag test inputs using embedding similarity to anchors (no LLM)
8
+ """
9
+ from __future__ import annotations
10
+ import json
11
+ import re
12
+ from typing import Optional
13
+ from sentence_transformers import SentenceTransformer
14
+ from sklearn.metrics.pairwise import cosine_similarity
15
+ import numpy as np
16
+
17
+ from diffprompt.models.cascade import call_cascade
18
+
19
+
20
+ INFER_PROMPT = """You are designing a test suite for an LLM prompt.
21
+
22
+ Prompt: {prompt}
23
+
24
+ Identify 3-4 dimensions that would reveal BEHAVIORAL differences in how this prompt responds.
25
+ Focus on: user intent, emotional state, topic complexity, and request type.
26
+ Avoid surface-level dimensions like length, format, or punctuation style.
27
+
28
+ Return ONLY valid JSON. Example:
29
+ {{
30
+ "tone": ["formal", "casual", "emotional", "urgent"],
31
+ "complexity": ["simple", "multi-part", "ambiguous"],
32
+ "intent": ["lookup", "reasoning", "emotional-support"]
33
+ }}
34
+
35
+ Each dimension must predict a meaningfully different response from this specific prompt."""
36
+
37
+
38
+ ANCHOR_PROMPT = """For the tag "{tag}" in dimension "{dimension}", write 3 short example sentences
39
+ that real users would send to this prompt: {prompt}
40
+
41
+ Each sentence must clearly represent "{tag}" and be distinct from each other.
42
+ Return ONLY a JSON array of 3 strings. No explanation."""
43
+
44
+
45
+ def _extract_json_object(raw: str) -> str:
46
+ """Extract the first {...} block from raw LLM output."""
47
+ clean = re.sub(r"```(?:json)?|```", "", raw).strip()
48
+ match = re.search(r'\{.*\}', clean, re.DOTALL)
49
+ return match.group(0) if match else clean
50
+
51
+
52
+ def _extract_json_array(raw: str) -> str:
53
+ """Extract the first [...] block from raw LLM output."""
54
+ clean = re.sub(r"```(?:json)?|```", "", raw).strip()
55
+ match = re.search(r'\[.*\]', clean, re.DOTALL)
56
+ return match.group(0) if match else clean
57
+
58
+
59
+ class Ontology:
60
+ def __init__(self):
61
+ self.dimensions: dict[str, list[str]] = {}
62
+ self.anchors: dict[str, dict[str, list[str]]] = {}
63
+ self.anchor_embeddings: dict[str, dict[str, np.ndarray]] = {}
64
+ self._embedder: Optional[SentenceTransformer] = None
65
+
66
+ @property
67
+ def embedder(self) -> SentenceTransformer:
68
+ if self._embedder is None:
69
+ import io, contextlib
70
+ f = io.StringIO()
71
+ with contextlib.redirect_stdout(f), contextlib.redirect_stderr(f):
72
+ self._embedder = SentenceTransformer("all-MiniLM-L6-v2")
73
+ return self._embedder
74
+
75
+ async def infer(self, prompt: str, local_only: bool = False) -> None:
76
+ """Infer dimensions and tags from the prompt. One LLM call."""
77
+ raw, _ = await call_cascade(
78
+ INFER_PROMPT.format(prompt=prompt),
79
+ local_only=local_only,
80
+ )
81
+ try:
82
+ parsed = json.loads(_extract_json_object(raw))
83
+ if len(parsed) == 1 and isinstance(list(parsed.values())[0], dict):
84
+ parsed = list(parsed.values())[0]
85
+ self.dimensions = {k: v for k, v in parsed.items() if isinstance(v, list)}
86
+ except json.JSONDecodeError:
87
+ self.dimensions = {
88
+ "user_intent": ["informational", "transactional", "complaint", "other"],
89
+ "emotional_state": ["neutral", "frustrated", "confused", "urgent"],
90
+ "request_type": ["specific", "open_ended", "troubleshooting"],
91
+ }
92
+
93
+ async def build_anchors(self, prompt: str, local_only: bool = False) -> None:
94
+ """Generate 3 anchor sentences per tag. One LLM call per tag."""
95
+ for dimension, tags in self.dimensions.items():
96
+ self.anchors[dimension] = {}
97
+ for tag in tags:
98
+ raw, _ = await call_cascade(
99
+ ANCHOR_PROMPT.format(tag=tag, dimension=dimension, prompt=prompt),
100
+ local_only=local_only,
101
+ )
102
+ try:
103
+ parsed = json.loads(_extract_json_array(raw))
104
+ if isinstance(parsed, list) and parsed:
105
+ self.anchors[dimension][tag] = [str(s) for s in parsed]
106
+ else:
107
+ self.anchors[dimension][tag] = [raw.strip()[:200]]
108
+ except Exception:
109
+ self.anchors[dimension][tag] = [raw.strip()[:200]]
110
+
111
+ # Build embeddings for all anchors
112
+ for dimension, tag_anchors in self.anchors.items():
113
+ self.anchor_embeddings[dimension] = {}
114
+ for tag, sentences in tag_anchors.items():
115
+ self.anchor_embeddings[dimension][tag] = self.embedder.encode(sentences)
116
+
117
+ def tag(self, input_text: str) -> dict[str, str]:
118
+ """
119
+ Tag an input using max similarity across multiple anchors per tag.
120
+ No LLM call — pure embeddings. Deterministic and fast.
121
+ """
122
+ if not self.anchor_embeddings:
123
+ return {}
124
+
125
+ input_emb = self.embedder.encode([input_text])
126
+ result = {}
127
+
128
+ for dimension, tags in self.dimensions.items():
129
+ if dimension not in self.anchor_embeddings:
130
+ continue
131
+ best_tag = tags[0]
132
+ best_score = -1.0
133
+
134
+ for tag in tags:
135
+ if tag not in self.anchor_embeddings[dimension]:
136
+ continue
137
+ anchor_embs = self.anchor_embeddings[dimension][tag]
138
+ sims = cosine_similarity(input_emb, anchor_embs)[0]
139
+ score = float(np.max(sims))
140
+ if score > best_score:
141
+ best_score = score
142
+ best_tag = tag
143
+
144
+ result[dimension] = best_tag
145
+
146
+ return result
147
+
148
+ def to_dict(self) -> dict:
149
+ return {"dimensions": self.dimensions, "anchors": self.anchors}
150
+
151
+ @classmethod
152
+ def from_dict(cls, data: dict) -> "Ontology":
153
+ o = cls()
154
+ o.dimensions = data["dimensions"]
155
+ o.anchors = data["anchors"]
156
+ for dimension, tag_anchors in o.anchors.items():
157
+ o.anchor_embeddings[dimension] = {}
158
+ for tag, sentences in tag_anchors.items():
159
+ o.anchor_embeddings[dimension][tag] = o.embedder.encode(sentences)
160
+ return o
@@ -0,0 +1,65 @@
1
+ """
2
+ Runs both prompts on all test cases.
3
+ Returns RunResult for each (test_case, prompt_version) pair.
4
+ """
5
+ from __future__ import annotations
6
+ import asyncio
7
+ import time
8
+ from diffprompt.models import TestCase, RunResult
9
+ from diffprompt.models.cascade import call_cascade
10
+
11
+
12
+ async def run_single(
13
+ test_case: TestCase,
14
+ prompt: str,
15
+ version: str,
16
+ model: str,
17
+ local_only: bool = False,
18
+ ) -> RunResult:
19
+ start = time.monotonic()
20
+ output, model_used = await call_cascade(
21
+ test_case.input,
22
+ system=prompt,
23
+ local_only=local_only,
24
+ )
25
+ latency_ms = (time.monotonic() - start) * 1000
26
+
27
+ return RunResult(
28
+ test_id=test_case.id,
29
+ prompt_version=version,
30
+ output=output,
31
+ model_used=model_used,
32
+ latency_ms=latency_ms,
33
+ )
34
+
35
+
36
+ async def run_both(
37
+ test_cases: list[TestCase],
38
+ prompt_v1: str,
39
+ prompt_v2: str,
40
+ model: str = "groq/llama-3.3-70b-versatile",
41
+ local_only: bool = False,
42
+ concurrency: int = 5,
43
+ ) -> tuple[dict[str, RunResult], dict[str, RunResult]]:
44
+ """
45
+ Run both prompts on all test cases concurrently.
46
+ Returns (v1_results, v2_results) as dicts keyed by test_id.
47
+ """
48
+ semaphore = asyncio.Semaphore(concurrency)
49
+
50
+ async def run_with_sem(tc, prompt, version):
51
+ async with semaphore:
52
+ return await run_single(tc, prompt, version, model, local_only)
53
+
54
+ # Fire all tasks concurrently (bounded by semaphore)
55
+ v1_tasks = [run_with_sem(tc, prompt_v1, "v1") for tc in test_cases]
56
+ v2_tasks = [run_with_sem(tc, prompt_v2, "v2") for tc in test_cases]
57
+
58
+ v1_raw, v2_raw = await asyncio.gather(
59
+ asyncio.gather(*v1_tasks),
60
+ asyncio.gather(*v2_tasks),
61
+ )
62
+
63
+ v1_results = {r.test_id: r for r in v1_raw}
64
+ v2_results = {r.test_id: r for r in v2_raw}
65
+ return v1_results, v2_results
@@ -0,0 +1,102 @@
1
+ """
2
+ Regression scoring and key example selection.
3
+ """
4
+ from __future__ import annotations
5
+ import numpy as np
6
+ from diffprompt.models import DiffResult, KeyExample, Verdict
7
+ from diffprompt.models.cascade import call_cascade
8
+
9
+
10
+ WHY_IT_MATTERS_PROMPT = """
11
+ Input: {input}
12
+ V1 output: {v1}
13
+ V2 output: {v2}
14
+ Verdict: {verdict}
15
+
16
+ In ONE sentence, explain what this reveals about the prompt change.
17
+ Focus on the mechanism — why did v2 behave differently here?
18
+ """
19
+
20
+
21
+ def regression_score(diffs: list[DiffResult]) -> float:
22
+ if not diffs:
23
+ return 50.0
24
+ total_weight = sum(d.divergence for d in diffs)
25
+ if total_weight == 0:
26
+ return 100.0
27
+ weighted_sum = sum(
28
+ d.divergence * (1.0 if d.verdict == Verdict.IMPROVEMENT else -1.0 if d.verdict == Verdict.REGRESSION else 0.0)
29
+ for d in diffs
30
+ )
31
+ return float(round(((weighted_sum / total_weight) + 1) / 2 * 100, 1))
32
+
33
+
34
+ def importance_score(diff: DiffResult) -> float:
35
+ divergence = diff.divergence
36
+ centrality = diff.cluster_centrality
37
+ input_length = len(diff.test_case.input.split())
38
+ simplicity = max(0.0, 1.0 - input_length / 50)
39
+ surprise = divergence * simplicity
40
+ return 0.4 * divergence + 0.3 * centrality + 0.3 * surprise
41
+
42
+
43
+ async def select_key_examples(
44
+ diffs: list[DiffResult],
45
+ top_n: int = 3,
46
+ local_only: bool = False,
47
+ ) -> list[KeyExample]:
48
+ if not diffs:
49
+ return []
50
+
51
+ for d in diffs:
52
+ d.importance_score = importance_score(d)
53
+
54
+ selected: list[tuple[str, DiffResult]] = []
55
+ used_ids: set[str] = set()
56
+
57
+ def _pick(slot: str, candidate) -> None:
58
+ if candidate and candidate.test_case.id not in used_ids:
59
+ selected.append((slot, candidate))
60
+ used_ids.add(candidate.test_case.id)
61
+
62
+ _pick("most_important", max(diffs, key=lambda d: d.importance_score))
63
+
64
+ improvements = [d for d in diffs if d.verdict == Verdict.IMPROVEMENT and d.test_case.id not in used_ids]
65
+ if improvements:
66
+ _pick("best_improvement", max(improvements, key=lambda d: d.divergence))
67
+
68
+ remaining = [d for d in diffs if d.test_case.id not in used_ids]
69
+ if remaining:
70
+ _pick("most_surprising", max(
71
+ remaining,
72
+ key=lambda d: d.divergence * max(0, 1 - len(d.test_case.input.split()) / 50),
73
+ ))
74
+
75
+ if top_n > 3:
76
+ regressions = sorted(
77
+ [d for d in diffs if d.verdict == Verdict.REGRESSION and d.test_case.id not in used_ids],
78
+ key=lambda d: d.divergence, reverse=True,
79
+ )
80
+ for d in regressions[:top_n - 3]:
81
+ _pick(f"regression_{len(selected)}", d)
82
+
83
+ selected = selected[:top_n]
84
+
85
+ examples = []
86
+ for slot_name, diff in selected:
87
+ why, _ = await call_cascade(
88
+ WHY_IT_MATTERS_PROMPT.format(
89
+ input=diff.test_case.input,
90
+ v1=diff.v1_output[:600],
91
+ v2=diff.v2_output[:600],
92
+ verdict=diff.verdict.value,
93
+ ),
94
+ local_only=local_only,
95
+ )
96
+ examples.append(KeyExample(
97
+ slot=slot_name,
98
+ diff=diff,
99
+ why_it_matters=why.strip(),
100
+ ))
101
+
102
+ return examples
@@ -0,0 +1,152 @@
1
+ """
2
+ Behavioral slicing.
3
+ Groups diffs by input tags and computes per-slice performance.
4
+ Recursive splitting for high-variance slices (max depth 3).
5
+ """
6
+ from __future__ import annotations
7
+ import numpy as np
8
+ from diffprompt.models import DiffResult, SliceResult, Verdict, TestCategory
9
+
10
+
11
+ MIN_SLICE_SIZE = 5
12
+ MAX_DEPTH = 3
13
+ VARIANCE_THRESHOLD = 0.1 # stop splitting if variance is already low
14
+ HIGH_VARIANCE_THRESHOLD = 0.15 # trigger recursive split above this
15
+
16
+
17
+ def compute_slices(diffs: list[DiffResult]) -> list[SliceResult]:
18
+ """Compute behavioral slices across all tag dimensions."""
19
+ if not diffs or not diffs[0].test_case.tags:
20
+ return []
21
+
22
+ dimensions = list(diffs[0].test_case.tags.keys())
23
+ slices = []
24
+
25
+ for dimension in dimensions:
26
+ # Group by tag value
27
+ groups: dict[str, list[DiffResult]] = {}
28
+ for d in diffs:
29
+ val = d.test_case.tags.get(dimension, "unknown")
30
+ groups.setdefault(val, []).append(d)
31
+
32
+ for value, group_diffs in groups.items():
33
+ slice_result = _compute_slice(
34
+ dimension=dimension,
35
+ value=value,
36
+ diffs=group_diffs,
37
+ depth=1,
38
+ )
39
+ if slice_result:
40
+ slices.append(slice_result)
41
+
42
+ # Recursive split if high variance + enough data + not too deep
43
+ if (
44
+ slice_result
45
+ and slice_result.variance > HIGH_VARIANCE_THRESHOLD
46
+ and len(group_diffs) >= MIN_SLICE_SIZE * 2
47
+ and slice_result.depth < MAX_DEPTH
48
+ ):
49
+ sub_slices = _recursive_split(group_diffs, dimension, depth=2)
50
+ slices.extend(sub_slices)
51
+
52
+ # Sort by mean_similarity ascending (worst slices first)
53
+ slices.sort(key=lambda s: s.mean_similarity)
54
+ return slices
55
+
56
+
57
+ def _compute_slice(
58
+ dimension: str,
59
+ value: str,
60
+ diffs: list[DiffResult],
61
+ depth: int,
62
+ ) -> SliceResult | None:
63
+ if len(diffs) < 2:
64
+ return None
65
+
66
+ sims = [d.similarity for d in diffs]
67
+ mean_sim = float(np.mean(sims))
68
+ variance = float(np.var(sims))
69
+
70
+ typical_ratio = sum(
71
+ 1 for d in diffs if d.test_case.category == TestCategory.TYPICAL
72
+ ) / len(diffs)
73
+
74
+ confidence = _compute_confidence(
75
+ n=len(diffs),
76
+ variance=variance,
77
+ typical_ratio=typical_ratio,
78
+ )
79
+
80
+ verdicts = [d.verdict for d in diffs]
81
+ verdict = _aggregate_verdict(verdicts)
82
+
83
+ return SliceResult(
84
+ dimension=dimension,
85
+ value=value,
86
+ label=f"{dimension}:{value}",
87
+ n=len(diffs),
88
+ mean_similarity=mean_sim,
89
+ variance=variance,
90
+ typical_ratio=typical_ratio,
91
+ confidence=confidence,
92
+ verdict=verdict,
93
+ depth=depth,
94
+ )
95
+
96
+
97
+ def _recursive_split(
98
+ diffs: list[DiffResult],
99
+ parent_dimension: str,
100
+ depth: int,
101
+ ) -> list[SliceResult]:
102
+ """Try splitting a high-variance slice by other dimensions."""
103
+ if depth > MAX_DEPTH or not diffs:
104
+ return []
105
+
106
+ other_dimensions = [
107
+ k for k in diffs[0].test_case.tags.keys()
108
+ if k != parent_dimension
109
+ ]
110
+
111
+ results = []
112
+ for dimension in other_dimensions:
113
+ groups: dict[str, list[DiffResult]] = {}
114
+ for d in diffs:
115
+ val = d.test_case.tags.get(dimension, "unknown")
116
+ groups.setdefault(val, []).append(d)
117
+
118
+ for value, group in groups.items():
119
+ if len(group) < MIN_SLICE_SIZE:
120
+ continue
121
+ slice_result = _compute_slice(
122
+ dimension=f"{parent_dimension}+{dimension}",
123
+ value=value,
124
+ diffs=group,
125
+ depth=depth,
126
+ )
127
+ if slice_result:
128
+ results.append(slice_result)
129
+
130
+ return results
131
+
132
+
133
+ def _compute_confidence(n: int, variance: float, typical_ratio: float) -> float:
134
+ """
135
+ Confidence = f(variance, typical_ratio, n).
136
+ Variance is checked first — high variance = impure slice = low confidence.
137
+ """
138
+ # Variance penalty (primary signal)
139
+ variance_score = max(0.0, 1.0 - variance * 5)
140
+
141
+ # Typical ratio bonus (real-world signal)
142
+ typical_score = typical_ratio
143
+
144
+ # N score (statistical power)
145
+ n_score = min(1.0, n / 20)
146
+
147
+ return float(0.5 * variance_score + 0.3 * typical_score + 0.2 * n_score)
148
+
149
+
150
+ def _aggregate_verdict(verdicts: list[Verdict]) -> Verdict:
151
+ counts = {v: verdicts.count(v) for v in Verdict}
152
+ return max(counts, key=counts.get)
@@ -0,0 +1,105 @@
1
+ """
2
+ Core data models for diffprompt.
3
+ These types flow through the entire pipeline — generator → runner → diff → analysis → output.
4
+ """
5
+ from __future__ import annotations
6
+ from enum import Enum
7
+ from typing import Optional
8
+ from pydantic import BaseModel, Field
9
+
10
+
11
+ class TestCategory(str, Enum):
12
+ TYPICAL = "typical"
13
+ BOUNDARY = "boundary"
14
+ ADVERSARIAL = "adversarial"
15
+ FORMAT = "format"
16
+
17
+
18
+ class Verdict(str, Enum):
19
+ IMPROVEMENT = "improvement"
20
+ REGRESSION = "regression"
21
+ NEUTRAL = "neutral"
22
+
23
+
24
+ class OutputFormat(str, Enum):
25
+ TERMINAL = "terminal"
26
+ JSON = "json"
27
+ HTML = "html"
28
+ TXT = "txt"
29
+
30
+
31
+ class TestCase(BaseModel):
32
+ id: str
33
+ input: str
34
+ category: TestCategory
35
+ tags: dict[str, str] = Field(default_factory=dict)
36
+
37
+
38
+ class RunResult(BaseModel):
39
+ test_id: str
40
+ prompt_version: str
41
+ output: str
42
+ model_used: str
43
+ latency_ms: Optional[float] = None
44
+
45
+
46
+ class DiffResult(BaseModel):
47
+ test_case: TestCase
48
+ v1_output: str
49
+ v2_output: str
50
+ similarity: float
51
+ divergence: float
52
+ verdict: Verdict
53
+ reason: str
54
+ judge_confidence: float
55
+ importance_score: float = 0.0
56
+ cluster_label: int = -1
57
+ cluster_centrality: float = 0.0
58
+
59
+
60
+ class SliceResult(BaseModel):
61
+ dimension: str
62
+ value: str
63
+ label: str
64
+ n: int
65
+ mean_similarity: float
66
+ variance: float
67
+ typical_ratio: float
68
+ confidence: float
69
+ verdict: Verdict
70
+ depth: int = 1
71
+
72
+
73
+ class Cluster(BaseModel):
74
+ label: int
75
+ name: str
76
+ description: str
77
+ n: int
78
+ mean_similarity: float
79
+ test_ids: list[str]
80
+
81
+
82
+ class KeyExample(BaseModel):
83
+ slot: str
84
+ diff: DiffResult
85
+ why_it_matters: str
86
+
87
+
88
+ class DiffReport(BaseModel):
89
+ prompt_v1: str
90
+ prompt_v2: str
91
+ model: str
92
+ judge: str
93
+ test_cases: list[TestCase]
94
+ diversity_score: float
95
+ diffs: list[DiffResult]
96
+ slices: list[SliceResult]
97
+ clusters: list[Cluster]
98
+ unclustered: list[DiffResult]
99
+ key_examples: list[KeyExample]
100
+ regression_score: float
101
+ n_improved: int
102
+ n_regressed: int
103
+ n_neutral: int
104
+ verdict: Verdict
105
+ recommendation: str