diffprompt 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- diffprompt-0.1.0/.gitignore +10 -0
- diffprompt-0.1.0/API_REFERENCE.md +548 -0
- diffprompt-0.1.0/CONTRIBUTING.md +209 -0
- diffprompt-0.1.0/COOKBOOK.md +290 -0
- diffprompt-0.1.0/PKG-INFO +249 -0
- diffprompt-0.1.0/README.md +219 -0
- diffprompt-0.1.0/diffprompt/__init__.py +2 -0
- diffprompt-0.1.0/diffprompt/cli.py +246 -0
- diffprompt-0.1.0/diffprompt/core/__init__.py +0 -0
- diffprompt-0.1.0/diffprompt/core/clusterer.py +130 -0
- diffprompt-0.1.0/diffprompt/core/embedder.py +63 -0
- diffprompt-0.1.0/diffprompt/core/generator.py +123 -0
- diffprompt-0.1.0/diffprompt/core/judge.py +94 -0
- diffprompt-0.1.0/diffprompt/core/ontology.py +160 -0
- diffprompt-0.1.0/diffprompt/core/runner.py +65 -0
- diffprompt-0.1.0/diffprompt/core/scorer.py +102 -0
- diffprompt-0.1.0/diffprompt/core/slicer.py +152 -0
- diffprompt-0.1.0/diffprompt/models/__init__.py +105 -0
- diffprompt-0.1.0/diffprompt/models/cascade.py +144 -0
- diffprompt-0.1.0/diffprompt/output/__init__.py +0 -0
- diffprompt-0.1.0/diffprompt/output/exporter.py +227 -0
- diffprompt-0.1.0/diffprompt/output/terminal.py +154 -0
- diffprompt-0.1.0/examples/basic_diff.py +98 -0
- diffprompt-0.1.0/examples/ontology_inspect.py +56 -0
- diffprompt-0.1.0/examples/similarity_playground.py +58 -0
- diffprompt-0.1.0/pyproject.toml +58 -0
- diffprompt-0.1.0/report.html +197 -0
- diffprompt-0.1.0/tests/__init__.py +0 -0
- diffprompt-0.1.0/tests/test_embedder.py +76 -0
- diffprompt-0.1.0/tests/test_generator.py +80 -0
- diffprompt-0.1.0/tests/test_slicer.py +93 -0
- diffprompt-0.1.0/v1.txt +1 -0
- diffprompt-0.1.0/v2.txt +1 -0
|
@@ -0,0 +1,548 @@
|
|
|
1
|
+
# API Reference
|
|
2
|
+
|
|
3
|
+
This document covers every public function and class in diffprompt. All types are Pydantic models unless noted otherwise.
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## Data models
|
|
8
|
+
|
|
9
|
+
All models live in `diffprompt.models`.
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
### TestCategory
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
class TestCategory(str, Enum):
|
|
17
|
+
TYPICAL = "typical"
|
|
18
|
+
BOUNDARY = "boundary"
|
|
19
|
+
ADVERSARIAL = "adversarial"
|
|
20
|
+
FORMAT = "format"
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
The four taxonomy buckets used for test generation.
|
|
24
|
+
|
|
25
|
+
- `TYPICAL` — realistic everyday inputs representing actual usage
|
|
26
|
+
- `BOUNDARY` — inputs at the edge of what the prompt handles; too long, too short, tangentially related
|
|
27
|
+
- `ADVERSARIAL` — inputs designed to expose failures; ambiguous, contradictory, trick questions
|
|
28
|
+
- `FORMAT` — unusual formatting; ALL CAPS, no punctuation, emojis, very short inputs
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
### Verdict
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
class Verdict(str, Enum):
|
|
36
|
+
IMPROVEMENT = "improvement"
|
|
37
|
+
REGRESSION = "regression"
|
|
38
|
+
NEUTRAL = "neutral"
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
The judgment produced by the judge LLM for each output pair.
|
|
42
|
+
|
|
43
|
+
---
|
|
44
|
+
|
|
45
|
+
### TestCase
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
class TestCase(BaseModel):
|
|
49
|
+
id: str
|
|
50
|
+
input: str
|
|
51
|
+
category: TestCategory
|
|
52
|
+
tags: dict[str, str]
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
A single test input. Created by `generate_test_cases()`, tagged by `Ontology.tag()`.
|
|
56
|
+
|
|
57
|
+
- `id` — short unique identifier (8 characters)
|
|
58
|
+
- `input` — the raw input string sent to the LLM
|
|
59
|
+
- `category` — which taxonomy bucket this came from
|
|
60
|
+
- `tags` — dimension-to-value mapping; e.g. `{"tone": "emotional", "complexity": "simple"}`
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
### RunResult
|
|
65
|
+
|
|
66
|
+
```python
|
|
67
|
+
class RunResult(BaseModel):
|
|
68
|
+
test_id: str
|
|
69
|
+
prompt_version: str
|
|
70
|
+
output: str
|
|
71
|
+
model_used: str
|
|
72
|
+
latency_ms: float | None
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
The output from running one test case through one prompt version.
|
|
76
|
+
|
|
77
|
+
- `test_id` — matches `TestCase.id`; used for joining v1 and v2 results
|
|
78
|
+
- `prompt_version` — `"v1"` or `"v2"`
|
|
79
|
+
- `model_used` — the model that produced this output; e.g. `"groq/llama-3.3-70b-versatile"`
|
|
80
|
+
- `latency_ms` — time in milliseconds for the LLM call
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
### DiffResult
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
class DiffResult(BaseModel):
|
|
88
|
+
test_case: TestCase
|
|
89
|
+
v1_output: str
|
|
90
|
+
v2_output: str
|
|
91
|
+
similarity: float
|
|
92
|
+
divergence: float
|
|
93
|
+
verdict: Verdict
|
|
94
|
+
reason: str
|
|
95
|
+
judge_confidence: float
|
|
96
|
+
importance_score: float
|
|
97
|
+
cluster_label: int
|
|
98
|
+
cluster_centrality: float
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
The core analysis unit. One per test case. Assembled in `cli.py` from embedder and judge outputs.
|
|
102
|
+
|
|
103
|
+
- `similarity` — cosine similarity between v1 and v2 outputs; 0.0 to 1.0
|
|
104
|
+
- `divergence` — `1 - similarity`; how much the outputs differ
|
|
105
|
+
- `verdict` — improvement, regression, or neutral
|
|
106
|
+
- `reason` — one-sentence explanation from the judge LLM
|
|
107
|
+
- `judge_confidence` — how certain the judge was; 0.0 to 1.0
|
|
108
|
+
- `importance_score` — computed by `scorer.importance_score()`; used for ranking key examples
|
|
109
|
+
- `cluster_label` — assigned by `cluster_diffs()`; -1 means unclustered noise
|
|
110
|
+
- `cluster_centrality` — how central this diff is within its cluster; populated by clusterer
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
### SliceResult
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
class SliceResult(BaseModel):
|
|
118
|
+
dimension: str
|
|
119
|
+
value: str
|
|
120
|
+
label: str
|
|
121
|
+
n: int
|
|
122
|
+
mean_similarity: float
|
|
123
|
+
variance: float
|
|
124
|
+
typical_ratio: float
|
|
125
|
+
confidence: float
|
|
126
|
+
verdict: Verdict
|
|
127
|
+
depth: int
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Performance summary for one behavioral slice.
|
|
131
|
+
|
|
132
|
+
- `dimension` — the tag dimension; e.g. `"tone"`
|
|
133
|
+
- `value` — the tag value; e.g. `"emotional"`
|
|
134
|
+
- `label` — `"{dimension}:{value}"`; e.g. `"tone:emotional"`
|
|
135
|
+
- `n` — number of test cases in this slice
|
|
136
|
+
- `mean_similarity` — average similarity across all diffs in this slice
|
|
137
|
+
- `variance` — variance of similarity scores; high variance means inconsistent behavior
|
|
138
|
+
- `typical_ratio` — fraction of diffs from `TYPICAL` bucket; affects confidence
|
|
139
|
+
- `confidence` — reliability of this slice's verdict; computed from variance, typical_ratio, and n
|
|
140
|
+
- `depth` — 1 for top-level slices; 2 or 3 for recursively split sub-slices
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
### Cluster
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
class Cluster(BaseModel):
|
|
148
|
+
label: int
|
|
149
|
+
name: str
|
|
150
|
+
description: str
|
|
151
|
+
n: int
|
|
152
|
+
mean_similarity: float
|
|
153
|
+
test_ids: list[str]
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
A named failure mode; a group of diffs with similar judge reasons.
|
|
157
|
+
|
|
158
|
+
- `label` — HDBSCAN cluster label (integer)
|
|
159
|
+
- `name` — auto-generated name; e.g. `"CONTEXT_LOSS"`, `"TONE_SHIFT"`, `"REFUSAL_SHIFT"`
|
|
160
|
+
- `description` — first 120 characters of the combined reasons from the top 3 diffs
|
|
161
|
+
- `test_ids` — list of `TestCase.id` values in this cluster
|
|
162
|
+
|
|
163
|
+
---
|
|
164
|
+
|
|
165
|
+
### KeyExample
|
|
166
|
+
|
|
167
|
+
```python
|
|
168
|
+
class KeyExample(BaseModel):
|
|
169
|
+
slot: str
|
|
170
|
+
diff: DiffResult
|
|
171
|
+
why_it_matters: str
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
One of the three highlighted examples in the output.
|
|
175
|
+
|
|
176
|
+
- `slot` — `"most_important"`, `"best_improvement"`, or `"most_surprising"`
|
|
177
|
+
- `diff` — the full `DiffResult` for this example
|
|
178
|
+
- `why_it_matters` — one-sentence explanation generated by the scorer LLM
|
|
179
|
+
|
|
180
|
+
---
|
|
181
|
+
|
|
182
|
+
### DiffReport
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
class DiffReport(BaseModel):
|
|
186
|
+
prompt_v1: str
|
|
187
|
+
prompt_v2: str
|
|
188
|
+
model: str
|
|
189
|
+
judge: str
|
|
190
|
+
test_cases: list[TestCase]
|
|
191
|
+
diversity_score: float
|
|
192
|
+
diffs: list[DiffResult]
|
|
193
|
+
slices: list[SliceResult]
|
|
194
|
+
clusters: list[Cluster]
|
|
195
|
+
unclustered: list[DiffResult]
|
|
196
|
+
key_examples: list[KeyExample]
|
|
197
|
+
regression_score: float
|
|
198
|
+
n_improved: int
|
|
199
|
+
n_regressed: int
|
|
200
|
+
n_neutral: int
|
|
201
|
+
verdict: Verdict
|
|
202
|
+
recommendation: str
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
The final report. Contains everything produced by the pipeline. Passed to the output layer for rendering.
|
|
206
|
+
|
|
207
|
+
- `diversity_score` — how diverse the test suite is; 0.0 to 1.0. Below 0.4 triggers a warning.
|
|
208
|
+
- `regression_score` — overall score 0 to 100. 100 means v2 improves everywhere. 0 means v2 regresses everywhere.
|
|
209
|
+
- `recommendation` — plain English summary of what to do
|
|
210
|
+
|
|
211
|
+
---
|
|
212
|
+
|
|
213
|
+
## Core functions
|
|
214
|
+
|
|
215
|
+
---
|
|
216
|
+
|
|
217
|
+
### diffprompt.core.ontology
|
|
218
|
+
|
|
219
|
+
```python
|
|
220
|
+
class Ontology:
|
|
221
|
+
dimensions: dict[str, list[str]]
|
|
222
|
+
anchors: dict[str, dict[str, str]]
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Manages prompt-specific dimensions and input tagging.
|
|
226
|
+
|
|
227
|
+
---
|
|
228
|
+
|
|
229
|
+
#### Ontology.infer
|
|
230
|
+
|
|
231
|
+
```python
|
|
232
|
+
async def infer(self, prompt: str, local_only: bool = False) -> None
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Calls the LLM once to infer relevant dimensions for this prompt. Populates `self.dimensions`.
|
|
236
|
+
|
|
237
|
+
- `prompt` — the prompt to analyze
|
|
238
|
+
- `local_only` — if True, never call external APIs
|
|
239
|
+
|
|
240
|
+
---
|
|
241
|
+
|
|
242
|
+
#### Ontology.build_anchors
|
|
243
|
+
|
|
244
|
+
```python
|
|
245
|
+
async def build_anchors(self, prompt: str, local_only: bool = False) -> None
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
Builds anchor phrases for each tag using zero-shot label embedding. No LLM calls. Must be called after `infer()`.
|
|
249
|
+
|
|
250
|
+
---
|
|
251
|
+
|
|
252
|
+
#### Ontology.tag
|
|
253
|
+
|
|
254
|
+
```python
|
|
255
|
+
def tag(self, input_text: str) -> dict[str, str]
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
Tags a single input by comparing it to anchor phrase embeddings. Returns a dimension-to-value mapping. No LLM call; pure embedding comparison.
|
|
259
|
+
|
|
260
|
+
```python
|
|
261
|
+
ontology.tag("I've been feeling anxious lately")
|
|
262
|
+
# {"tone": "emotional", "complexity": "simple", "intent": "seeking-support"}
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
---
|
|
266
|
+
|
|
267
|
+
#### Ontology.to_dict / from_dict
|
|
268
|
+
|
|
269
|
+
```python
|
|
270
|
+
def to_dict(self) -> dict
|
|
271
|
+
@classmethod
|
|
272
|
+
def from_dict(cls, data: dict) -> Ontology
|
|
273
|
+
```
|
|
274
|
+
|
|
275
|
+
Serialize and deserialize the ontology for caching to disk.
|
|
276
|
+
|
|
277
|
+
---
|
|
278
|
+
|
|
279
|
+
### diffprompt.core.generator
|
|
280
|
+
|
|
281
|
+
#### generate_test_cases
|
|
282
|
+
|
|
283
|
+
```python
|
|
284
|
+
async def generate_test_cases(
|
|
285
|
+
prompt: str,
|
|
286
|
+
n: int = 40,
|
|
287
|
+
ontology: Ontology | None = None,
|
|
288
|
+
local_only: bool = False,
|
|
289
|
+
) -> list[TestCase]
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
Generates `n` test cases distributed across the four taxonomy buckets (45% typical, 35% adversarial, 10% boundary, 10% format). Each test case is tagged using the ontology if provided.
|
|
293
|
+
|
|
294
|
+
- `n` — total number of test cases to generate. Actual count may differ slightly due to rounding.
|
|
295
|
+
- `ontology` — if provided, tags each generated input. If None, tags are empty.
|
|
296
|
+
|
|
297
|
+
---
|
|
298
|
+
|
|
299
|
+
#### diversity_score
|
|
300
|
+
|
|
301
|
+
```python
|
|
302
|
+
def diversity_score(test_cases: list[TestCase], embedder: SentenceTransformer) -> float
|
|
303
|
+
```
|
|
304
|
+
|
|
305
|
+
Computes how diverse the test suite is. Returns `1 - mean_pairwise_similarity`. Higher is more diverse. A score below 0.4 means many inputs are semantically redundant.
|
|
306
|
+
|
|
307
|
+
---
|
|
308
|
+
|
|
309
|
+
### diffprompt.core.embedder
|
|
310
|
+
|
|
311
|
+
#### get_embedder
|
|
312
|
+
|
|
313
|
+
```python
|
|
314
|
+
@lru_cache(maxsize=1)
|
|
315
|
+
def get_embedder() -> SentenceTransformer
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
Returns the cached `all-MiniLM-L6-v2` model. Loads on first call, returns the same instance on all subsequent calls.
|
|
319
|
+
|
|
320
|
+
---
|
|
321
|
+
|
|
322
|
+
#### embed
|
|
323
|
+
|
|
324
|
+
```python
|
|
325
|
+
def embed(texts: list[str]) -> np.ndarray
|
|
326
|
+
```
|
|
327
|
+
|
|
328
|
+
Embeds a list of texts. Returns a matrix of shape `(len(texts), 384)`.
|
|
329
|
+
|
|
330
|
+
---
|
|
331
|
+
|
|
332
|
+
#### similarity
|
|
333
|
+
|
|
334
|
+
```python
|
|
335
|
+
def similarity(text_a: str, text_b: str) -> float
|
|
336
|
+
```
|
|
337
|
+
|
|
338
|
+
Cosine similarity between two texts. Returns a float between 0.0 and 1.0.
|
|
339
|
+
|
|
340
|
+
---
|
|
341
|
+
|
|
342
|
+
#### batch_similarity
|
|
343
|
+
|
|
344
|
+
```python
|
|
345
|
+
def batch_similarity(pairs: list[tuple[str, str]]) -> list[float]
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
Efficient similarity for many pairs. Embeds all texts in one pass. Returns one float per pair, in the same order as input.
|
|
349
|
+
|
|
350
|
+
```python
|
|
351
|
+
scores = batch_similarity([
|
|
352
|
+
("Paris is the capital of France", "France's capital is Paris"),
|
|
353
|
+
("Hello world", "Goodbye world"),
|
|
354
|
+
])
|
|
355
|
+
# [0.94, 0.41]
|
|
356
|
+
```
|
|
357
|
+
|
|
358
|
+
---
|
|
359
|
+
|
|
360
|
+
### diffprompt.core.runner
|
|
361
|
+
|
|
362
|
+
#### run_single
|
|
363
|
+
|
|
364
|
+
```python
|
|
365
|
+
async def run_single(
|
|
366
|
+
test_case: TestCase,
|
|
367
|
+
prompt: str,
|
|
368
|
+
version: str,
|
|
369
|
+
model: str,
|
|
370
|
+
local_only: bool = False,
|
|
371
|
+
) -> RunResult
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
Runs one test case through one prompt version. Returns a `RunResult` with the output and latency.
|
|
375
|
+
|
|
376
|
+
- `prompt` — used as the system message
|
|
377
|
+
- `test_case.input` — used as the user message
|
|
378
|
+
- `version` — label for the result; `"v1"` or `"v2"`
|
|
379
|
+
|
|
380
|
+
---
|
|
381
|
+
|
|
382
|
+
#### run_both
|
|
383
|
+
|
|
384
|
+
```python
|
|
385
|
+
async def run_both(
|
|
386
|
+
test_cases: list[TestCase],
|
|
387
|
+
prompt_v1: str,
|
|
388
|
+
prompt_v2: str,
|
|
389
|
+
model: str = "groq/llama-3.3-70b-versatile",
|
|
390
|
+
local_only: bool = False,
|
|
391
|
+
concurrency: int = 5,
|
|
392
|
+
) -> tuple[dict[str, RunResult], dict[str, RunResult]]
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
Runs all test cases through both prompts concurrently. Returns two dicts keyed by `test_id`.
|
|
396
|
+
|
|
397
|
+
- `concurrency` — max concurrent LLM calls. Increase for faster runs, decrease to avoid rate limits.
|
|
398
|
+
|
|
399
|
+
```python
|
|
400
|
+
v1_results, v2_results = await run_both(test_cases, prompt_v1, prompt_v2)
|
|
401
|
+
v1_output = v1_results["a3f8b2"].output
|
|
402
|
+
v2_output = v2_results["a3f8b2"].output
|
|
403
|
+
```
|
|
404
|
+
|
|
405
|
+
---
|
|
406
|
+
|
|
407
|
+
### diffprompt.core.judge
|
|
408
|
+
|
|
409
|
+
#### judge_single
|
|
410
|
+
|
|
411
|
+
```python
|
|
412
|
+
async def judge_single(
|
|
413
|
+
test_case: TestCase,
|
|
414
|
+
v1_output: str,
|
|
415
|
+
v2_output: str,
|
|
416
|
+
similarity: float,
|
|
417
|
+
local_only: bool = False,
|
|
418
|
+
) -> tuple[Verdict, str, float]
|
|
419
|
+
```
|
|
420
|
+
|
|
421
|
+
Judges one output pair. Returns `(verdict, reason, confidence)`.
|
|
422
|
+
|
|
423
|
+
Short-circuits to `(NEUTRAL, "outputs are semantically identical", 1.0)` if `similarity > 0.95`. Automatically escalates to the Groq 70B model if confidence is below `0.65`.
|
|
424
|
+
|
|
425
|
+
---
|
|
426
|
+
|
|
427
|
+
### diffprompt.core.clusterer
|
|
428
|
+
|
|
429
|
+
#### cluster_diffs
|
|
430
|
+
|
|
431
|
+
```python
|
|
432
|
+
def cluster_diffs(diffs: list[DiffResult]) -> tuple[list[Cluster], list[DiffResult]]
|
|
433
|
+
```
|
|
434
|
+
|
|
435
|
+
Clusters diffs by their judge reasons using HDBSCAN + UMAP. Returns `(clusters, unclustered)` where `unclustered` contains noise points (HDBSCAN label -1).
|
|
436
|
+
|
|
437
|
+
Requires `hdbscan` and `umap-learn` to be installed. Returns `([], diffs)` if fewer than 4 diffs are provided.
|
|
438
|
+
|
|
439
|
+
---
|
|
440
|
+
|
|
441
|
+
### diffprompt.core.slicer
|
|
442
|
+
|
|
443
|
+
#### compute_slices
|
|
444
|
+
|
|
445
|
+
```python
|
|
446
|
+
def compute_slices(diffs: list[DiffResult]) -> list[SliceResult]
|
|
447
|
+
```
|
|
448
|
+
|
|
449
|
+
Groups diffs by their tag dimensions and computes performance per slice. Returns slices sorted by `mean_similarity` ascending (worst first). Recursively splits high-variance slices up to depth 3.
|
|
450
|
+
|
|
451
|
+
Returns an empty list if diffs have no tags.
|
|
452
|
+
|
|
453
|
+
---
|
|
454
|
+
|
|
455
|
+
### diffprompt.core.scorer
|
|
456
|
+
|
|
457
|
+
#### regression_score
|
|
458
|
+
|
|
459
|
+
```python
|
|
460
|
+
def regression_score(diffs: list[DiffResult]) -> float
|
|
461
|
+
```
|
|
462
|
+
|
|
463
|
+
Computes overall regression score from 0 to 100. Weighted by divergence so large behavioral changes matter more than small ones. Returns 50.0 for empty input and 100.0 if all outputs are identical.
|
|
464
|
+
|
|
465
|
+
Formula: `((weighted_sum / total_weight) + 1) / 2 * 100`
|
|
466
|
+
where improvements contribute `+divergence` and regressions contribute `-divergence`.
|
|
467
|
+
|
|
468
|
+
---
|
|
469
|
+
|
|
470
|
+
#### importance_score
|
|
471
|
+
|
|
472
|
+
```python
|
|
473
|
+
def importance_score(diff: DiffResult) -> float
|
|
474
|
+
```
|
|
475
|
+
|
|
476
|
+
Ranks a single diff by how informative it is. Returns a float between 0 and 1.
|
|
477
|
+
|
|
478
|
+
Formula: `0.4 * divergence + 0.3 * cluster_centrality + 0.3 * surprise`
|
|
479
|
+
where `surprise = divergence * (1 - input_length / 50)`. Short inputs that changed a lot are surprising.
|
|
480
|
+
|
|
481
|
+
---
|
|
482
|
+
|
|
483
|
+
#### select_key_examples
|
|
484
|
+
|
|
485
|
+
```python
|
|
486
|
+
async def select_key_examples(
|
|
487
|
+
diffs: list[DiffResult],
|
|
488
|
+
local_only: bool = False,
|
|
489
|
+
) -> list[KeyExample]
|
|
490
|
+
```
|
|
491
|
+
|
|
492
|
+
Selects up to three key examples and generates a "why it matters" sentence for each via LLM call.
|
|
493
|
+
|
|
494
|
+
- Slot 1: Most Important (highest `importance_score`)
|
|
495
|
+
- Slot 2: Best Improvement (highest divergence among improvements; omitted if none)
|
|
496
|
+
- Slot 3: Most Surprising (high divergence on short input, excluding slot 1)
|
|
497
|
+
|
|
498
|
+
---
|
|
499
|
+
|
|
500
|
+
### diffprompt.models.cascade
|
|
501
|
+
|
|
502
|
+
#### call_cascade
|
|
503
|
+
|
|
504
|
+
```python
|
|
505
|
+
async def call_cascade(
|
|
506
|
+
prompt: str,
|
|
507
|
+
system: str | None = None,
|
|
508
|
+
local_model: str = "qwen2.5:7b",
|
|
509
|
+
groq_model: str = "llama-3.3-70b-versatile",
|
|
510
|
+
local_only: bool = False,
|
|
511
|
+
) -> tuple[str, str]
|
|
512
|
+
```
|
|
513
|
+
|
|
514
|
+
The main LLM entry point. Tries Ollama first, falls back to Groq. Returns `(output, model_used)`.
|
|
515
|
+
|
|
516
|
+
- `prompt` — the user message
|
|
517
|
+
- `system` — optional system message
|
|
518
|
+
- `local_only` — if True, raises `RuntimeError` when Ollama is unavailable instead of falling back to Groq
|
|
519
|
+
|
|
520
|
+
Raises `RuntimeError` if all models fail.
|
|
521
|
+
|
|
522
|
+
---
|
|
523
|
+
|
|
524
|
+
#### call_ollama
|
|
525
|
+
|
|
526
|
+
```python
|
|
527
|
+
async def call_ollama(
|
|
528
|
+
model: str,
|
|
529
|
+
prompt: str,
|
|
530
|
+
system: str | None = None,
|
|
531
|
+
) -> str | None
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
Calls a local Ollama model. Returns `None` (never raises) if Ollama is not running or the call fails.
|
|
535
|
+
|
|
536
|
+
---
|
|
537
|
+
|
|
538
|
+
#### call_groq
|
|
539
|
+
|
|
540
|
+
```python
|
|
541
|
+
async def call_groq(
|
|
542
|
+
model: str,
|
|
543
|
+
prompt: str,
|
|
544
|
+
system: str | None = None,
|
|
545
|
+
) -> str | None
|
|
546
|
+
```
|
|
547
|
+
|
|
548
|
+
Calls the Groq API. Returns `None` if `GROQ_API_KEY` is not set or the call fails. Reads the key from `os.getenv("GROQ_API_KEY")`.
|