diffprompt 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. diffprompt-0.1.0/.gitignore +10 -0
  2. diffprompt-0.1.0/API_REFERENCE.md +548 -0
  3. diffprompt-0.1.0/CONTRIBUTING.md +209 -0
  4. diffprompt-0.1.0/COOKBOOK.md +290 -0
  5. diffprompt-0.1.0/PKG-INFO +249 -0
  6. diffprompt-0.1.0/README.md +219 -0
  7. diffprompt-0.1.0/diffprompt/__init__.py +2 -0
  8. diffprompt-0.1.0/diffprompt/cli.py +246 -0
  9. diffprompt-0.1.0/diffprompt/core/__init__.py +0 -0
  10. diffprompt-0.1.0/diffprompt/core/clusterer.py +130 -0
  11. diffprompt-0.1.0/diffprompt/core/embedder.py +63 -0
  12. diffprompt-0.1.0/diffprompt/core/generator.py +123 -0
  13. diffprompt-0.1.0/diffprompt/core/judge.py +94 -0
  14. diffprompt-0.1.0/diffprompt/core/ontology.py +160 -0
  15. diffprompt-0.1.0/diffprompt/core/runner.py +65 -0
  16. diffprompt-0.1.0/diffprompt/core/scorer.py +102 -0
  17. diffprompt-0.1.0/diffprompt/core/slicer.py +152 -0
  18. diffprompt-0.1.0/diffprompt/models/__init__.py +105 -0
  19. diffprompt-0.1.0/diffprompt/models/cascade.py +144 -0
  20. diffprompt-0.1.0/diffprompt/output/__init__.py +0 -0
  21. diffprompt-0.1.0/diffprompt/output/exporter.py +227 -0
  22. diffprompt-0.1.0/diffprompt/output/terminal.py +154 -0
  23. diffprompt-0.1.0/examples/basic_diff.py +98 -0
  24. diffprompt-0.1.0/examples/ontology_inspect.py +56 -0
  25. diffprompt-0.1.0/examples/similarity_playground.py +58 -0
  26. diffprompt-0.1.0/pyproject.toml +58 -0
  27. diffprompt-0.1.0/report.html +197 -0
  28. diffprompt-0.1.0/tests/__init__.py +0 -0
  29. diffprompt-0.1.0/tests/test_embedder.py +76 -0
  30. diffprompt-0.1.0/tests/test_generator.py +80 -0
  31. diffprompt-0.1.0/tests/test_slicer.py +93 -0
  32. diffprompt-0.1.0/v1.txt +1 -0
  33. diffprompt-0.1.0/v2.txt +1 -0
@@ -0,0 +1,10 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ dist/
5
+ build/
6
+ .env
7
+ .venv/
8
+ venv/
9
+ diffprompt_report.*
10
+ diffprompt.ontology.json
@@ -0,0 +1,548 @@
1
+ # API Reference
2
+
3
+ This document covers every public function and class in diffprompt. All types are Pydantic models unless noted otherwise.
4
+
5
+ ---
6
+
7
+ ## Data models
8
+
9
+ All models live in `diffprompt.models`.
10
+
11
+ ---
12
+
13
+ ### TestCategory
14
+
15
+ ```python
16
+ class TestCategory(str, Enum):
17
+ TYPICAL = "typical"
18
+ BOUNDARY = "boundary"
19
+ ADVERSARIAL = "adversarial"
20
+ FORMAT = "format"
21
+ ```
22
+
23
+ The four taxonomy buckets used for test generation.
24
+
25
+ - `TYPICAL` — realistic everyday inputs representing actual usage
26
+ - `BOUNDARY` — inputs at the edge of what the prompt handles; too long, too short, tangentially related
27
+ - `ADVERSARIAL` — inputs designed to expose failures; ambiguous, contradictory, trick questions
28
+ - `FORMAT` — unusual formatting; ALL CAPS, no punctuation, emojis, very short inputs
29
+
30
+ ---
31
+
32
+ ### Verdict
33
+
34
+ ```python
35
+ class Verdict(str, Enum):
36
+ IMPROVEMENT = "improvement"
37
+ REGRESSION = "regression"
38
+ NEUTRAL = "neutral"
39
+ ```
40
+
41
+ The judgment produced by the judge LLM for each output pair.
42
+
43
+ ---
44
+
45
+ ### TestCase
46
+
47
+ ```python
48
+ class TestCase(BaseModel):
49
+ id: str
50
+ input: str
51
+ category: TestCategory
52
+ tags: dict[str, str]
53
+ ```
54
+
55
+ A single test input. Created by `generate_test_cases()`, tagged by `Ontology.tag()`.
56
+
57
+ - `id` — short unique identifier (8 characters)
58
+ - `input` — the raw input string sent to the LLM
59
+ - `category` — which taxonomy bucket this came from
60
+ - `tags` — dimension-to-value mapping; e.g. `{"tone": "emotional", "complexity": "simple"}`
61
+
62
+ ---
63
+
64
+ ### RunResult
65
+
66
+ ```python
67
+ class RunResult(BaseModel):
68
+ test_id: str
69
+ prompt_version: str
70
+ output: str
71
+ model_used: str
72
+ latency_ms: float | None
73
+ ```
74
+
75
+ The output from running one test case through one prompt version.
76
+
77
+ - `test_id` — matches `TestCase.id`; used for joining v1 and v2 results
78
+ - `prompt_version` — `"v1"` or `"v2"`
79
+ - `model_used` — the model that produced this output; e.g. `"groq/llama-3.3-70b-versatile"`
80
+ - `latency_ms` — time in milliseconds for the LLM call
81
+
82
+ ---
83
+
84
+ ### DiffResult
85
+
86
+ ```python
87
+ class DiffResult(BaseModel):
88
+ test_case: TestCase
89
+ v1_output: str
90
+ v2_output: str
91
+ similarity: float
92
+ divergence: float
93
+ verdict: Verdict
94
+ reason: str
95
+ judge_confidence: float
96
+ importance_score: float
97
+ cluster_label: int
98
+ cluster_centrality: float
99
+ ```
100
+
101
+ The core analysis unit. One per test case. Assembled in `cli.py` from embedder and judge outputs.
102
+
103
+ - `similarity` — cosine similarity between v1 and v2 outputs; 0.0 to 1.0
104
+ - `divergence` — `1 - similarity`; how much the outputs differ
105
+ - `verdict` — improvement, regression, or neutral
106
+ - `reason` — one-sentence explanation from the judge LLM
107
+ - `judge_confidence` — how certain the judge was; 0.0 to 1.0
108
+ - `importance_score` — computed by `scorer.importance_score()`; used for ranking key examples
109
+ - `cluster_label` — assigned by `cluster_diffs()`; -1 means unclustered noise
110
+ - `cluster_centrality` — how central this diff is within its cluster; populated by clusterer
111
+
112
+ ---
113
+
114
+ ### SliceResult
115
+
116
+ ```python
117
+ class SliceResult(BaseModel):
118
+ dimension: str
119
+ value: str
120
+ label: str
121
+ n: int
122
+ mean_similarity: float
123
+ variance: float
124
+ typical_ratio: float
125
+ confidence: float
126
+ verdict: Verdict
127
+ depth: int
128
+ ```
129
+
130
+ Performance summary for one behavioral slice.
131
+
132
+ - `dimension` — the tag dimension; e.g. `"tone"`
133
+ - `value` — the tag value; e.g. `"emotional"`
134
+ - `label` — `"{dimension}:{value}"`; e.g. `"tone:emotional"`
135
+ - `n` — number of test cases in this slice
136
+ - `mean_similarity` — average similarity across all diffs in this slice
137
+ - `variance` — variance of similarity scores; high variance means inconsistent behavior
138
+ - `typical_ratio` — fraction of diffs from `TYPICAL` bucket; affects confidence
139
+ - `confidence` — reliability of this slice's verdict; computed from variance, typical_ratio, and n
140
+ - `depth` — 1 for top-level slices; 2 or 3 for recursively split sub-slices
141
+
142
+ ---
143
+
144
+ ### Cluster
145
+
146
+ ```python
147
+ class Cluster(BaseModel):
148
+ label: int
149
+ name: str
150
+ description: str
151
+ n: int
152
+ mean_similarity: float
153
+ test_ids: list[str]
154
+ ```
155
+
156
+ A named failure mode; a group of diffs with similar judge reasons.
157
+
158
+ - `label` — HDBSCAN cluster label (integer)
159
+ - `name` — auto-generated name; e.g. `"CONTEXT_LOSS"`, `"TONE_SHIFT"`, `"REFUSAL_SHIFT"`
160
+ - `description` — first 120 characters of the combined reasons from the top 3 diffs
161
+ - `test_ids` — list of `TestCase.id` values in this cluster
162
+
163
+ ---
164
+
165
+ ### KeyExample
166
+
167
+ ```python
168
+ class KeyExample(BaseModel):
169
+ slot: str
170
+ diff: DiffResult
171
+ why_it_matters: str
172
+ ```
173
+
174
+ One of the three highlighted examples in the output.
175
+
176
+ - `slot` — `"most_important"`, `"best_improvement"`, or `"most_surprising"`
177
+ - `diff` — the full `DiffResult` for this example
178
+ - `why_it_matters` — one-sentence explanation generated by the scorer LLM
179
+
180
+ ---
181
+
182
+ ### DiffReport
183
+
184
+ ```python
185
+ class DiffReport(BaseModel):
186
+ prompt_v1: str
187
+ prompt_v2: str
188
+ model: str
189
+ judge: str
190
+ test_cases: list[TestCase]
191
+ diversity_score: float
192
+ diffs: list[DiffResult]
193
+ slices: list[SliceResult]
194
+ clusters: list[Cluster]
195
+ unclustered: list[DiffResult]
196
+ key_examples: list[KeyExample]
197
+ regression_score: float
198
+ n_improved: int
199
+ n_regressed: int
200
+ n_neutral: int
201
+ verdict: Verdict
202
+ recommendation: str
203
+ ```
204
+
205
+ The final report. Contains everything produced by the pipeline. Passed to the output layer for rendering.
206
+
207
+ - `diversity_score` — how diverse the test suite is; 0.0 to 1.0. Below 0.4 triggers a warning.
208
+ - `regression_score` — overall score 0 to 100. 100 means v2 improves everywhere. 0 means v2 regresses everywhere.
209
+ - `recommendation` — plain English summary of what to do
210
+
211
+ ---
212
+
213
+ ## Core functions
214
+
215
+ ---
216
+
217
+ ### diffprompt.core.ontology
218
+
219
+ ```python
220
+ class Ontology:
221
+ dimensions: dict[str, list[str]]
222
+ anchors: dict[str, dict[str, str]]
223
+ ```
224
+
225
+ Manages prompt-specific dimensions and input tagging.
226
+
227
+ ---
228
+
229
+ #### Ontology.infer
230
+
231
+ ```python
232
+ async def infer(self, prompt: str, local_only: bool = False) -> None
233
+ ```
234
+
235
+ Calls the LLM once to infer relevant dimensions for this prompt. Populates `self.dimensions`.
236
+
237
+ - `prompt` — the prompt to analyze
238
+ - `local_only` — if True, never call external APIs
239
+
240
+ ---
241
+
242
+ #### Ontology.build_anchors
243
+
244
+ ```python
245
+ async def build_anchors(self, prompt: str, local_only: bool = False) -> None
246
+ ```
247
+
248
+ Builds anchor phrases for each tag using zero-shot label embedding. No LLM calls. Must be called after `infer()`.
249
+
250
+ ---
251
+
252
+ #### Ontology.tag
253
+
254
+ ```python
255
+ def tag(self, input_text: str) -> dict[str, str]
256
+ ```
257
+
258
+ Tags a single input by comparing it to anchor phrase embeddings. Returns a dimension-to-value mapping. No LLM call; pure embedding comparison.
259
+
260
+ ```python
261
+ ontology.tag("I've been feeling anxious lately")
262
+ # {"tone": "emotional", "complexity": "simple", "intent": "seeking-support"}
263
+ ```
264
+
265
+ ---
266
+
267
+ #### Ontology.to_dict / from_dict
268
+
269
+ ```python
270
+ def to_dict(self) -> dict
271
+ @classmethod
272
+ def from_dict(cls, data: dict) -> Ontology
273
+ ```
274
+
275
+ Serialize and deserialize the ontology for caching to disk.
276
+
277
+ ---
278
+
279
+ ### diffprompt.core.generator
280
+
281
+ #### generate_test_cases
282
+
283
+ ```python
284
+ async def generate_test_cases(
285
+ prompt: str,
286
+ n: int = 40,
287
+ ontology: Ontology | None = None,
288
+ local_only: bool = False,
289
+ ) -> list[TestCase]
290
+ ```
291
+
292
+ Generates `n` test cases distributed across the four taxonomy buckets (45% typical, 35% adversarial, 10% boundary, 10% format). Each test case is tagged using the ontology if provided.
293
+
294
+ - `n` — total number of test cases to generate. Actual count may differ slightly due to rounding.
295
+ - `ontology` — if provided, tags each generated input. If None, tags are empty.
296
+
297
+ ---
298
+
299
+ #### diversity_score
300
+
301
+ ```python
302
+ def diversity_score(test_cases: list[TestCase], embedder: SentenceTransformer) -> float
303
+ ```
304
+
305
+ Computes how diverse the test suite is. Returns `1 - mean_pairwise_similarity`. Higher is more diverse. A score below 0.4 means many inputs are semantically redundant.
306
+
307
+ ---
308
+
309
+ ### diffprompt.core.embedder
310
+
311
+ #### get_embedder
312
+
313
+ ```python
314
+ @lru_cache(maxsize=1)
315
+ def get_embedder() -> SentenceTransformer
316
+ ```
317
+
318
+ Returns the cached `all-MiniLM-L6-v2` model. Loads on first call, returns the same instance on all subsequent calls.
319
+
320
+ ---
321
+
322
+ #### embed
323
+
324
+ ```python
325
+ def embed(texts: list[str]) -> np.ndarray
326
+ ```
327
+
328
+ Embeds a list of texts. Returns a matrix of shape `(len(texts), 384)`.
329
+
330
+ ---
331
+
332
+ #### similarity
333
+
334
+ ```python
335
+ def similarity(text_a: str, text_b: str) -> float
336
+ ```
337
+
338
+ Cosine similarity between two texts. Returns a float between 0.0 and 1.0.
339
+
340
+ ---
341
+
342
+ #### batch_similarity
343
+
344
+ ```python
345
+ def batch_similarity(pairs: list[tuple[str, str]]) -> list[float]
346
+ ```
347
+
348
+ Efficient similarity for many pairs. Embeds all texts in one pass. Returns one float per pair, in the same order as input.
349
+
350
+ ```python
351
+ scores = batch_similarity([
352
+ ("Paris is the capital of France", "France's capital is Paris"),
353
+ ("Hello world", "Goodbye world"),
354
+ ])
355
+ # [0.94, 0.41]
356
+ ```
357
+
358
+ ---
359
+
360
+ ### diffprompt.core.runner
361
+
362
+ #### run_single
363
+
364
+ ```python
365
+ async def run_single(
366
+ test_case: TestCase,
367
+ prompt: str,
368
+ version: str,
369
+ model: str,
370
+ local_only: bool = False,
371
+ ) -> RunResult
372
+ ```
373
+
374
+ Runs one test case through one prompt version. Returns a `RunResult` with the output and latency.
375
+
376
+ - `prompt` — used as the system message
377
+ - `test_case.input` — used as the user message
378
+ - `version` — label for the result; `"v1"` or `"v2"`
379
+
380
+ ---
381
+
382
+ #### run_both
383
+
384
+ ```python
385
+ async def run_both(
386
+ test_cases: list[TestCase],
387
+ prompt_v1: str,
388
+ prompt_v2: str,
389
+ model: str = "groq/llama-3.3-70b-versatile",
390
+ local_only: bool = False,
391
+ concurrency: int = 5,
392
+ ) -> tuple[dict[str, RunResult], dict[str, RunResult]]
393
+ ```
394
+
395
+ Runs all test cases through both prompts concurrently. Returns two dicts keyed by `test_id`.
396
+
397
+ - `concurrency` — max concurrent LLM calls. Increase for faster runs, decrease to avoid rate limits.
398
+
399
+ ```python
400
+ v1_results, v2_results = await run_both(test_cases, prompt_v1, prompt_v2)
401
+ v1_output = v1_results["a3f8b2"].output
402
+ v2_output = v2_results["a3f8b2"].output
403
+ ```
404
+
405
+ ---
406
+
407
+ ### diffprompt.core.judge
408
+
409
+ #### judge_single
410
+
411
+ ```python
412
+ async def judge_single(
413
+ test_case: TestCase,
414
+ v1_output: str,
415
+ v2_output: str,
416
+ similarity: float,
417
+ local_only: bool = False,
418
+ ) -> tuple[Verdict, str, float]
419
+ ```
420
+
421
+ Judges one output pair. Returns `(verdict, reason, confidence)`.
422
+
423
+ Short-circuits to `(NEUTRAL, "outputs are semantically identical", 1.0)` if `similarity > 0.95`. Automatically escalates to the Groq 70B model if confidence is below `0.65`.
424
+
425
+ ---
426
+
427
+ ### diffprompt.core.clusterer
428
+
429
+ #### cluster_diffs
430
+
431
+ ```python
432
+ def cluster_diffs(diffs: list[DiffResult]) -> tuple[list[Cluster], list[DiffResult]]
433
+ ```
434
+
435
+ Clusters diffs by their judge reasons using HDBSCAN + UMAP. Returns `(clusters, unclustered)` where `unclustered` contains noise points (HDBSCAN label -1).
436
+
437
+ Requires `hdbscan` and `umap-learn` to be installed. Returns `([], diffs)` if fewer than 4 diffs are provided.
438
+
439
+ ---
440
+
441
+ ### diffprompt.core.slicer
442
+
443
+ #### compute_slices
444
+
445
+ ```python
446
+ def compute_slices(diffs: list[DiffResult]) -> list[SliceResult]
447
+ ```
448
+
449
+ Groups diffs by their tag dimensions and computes performance per slice. Returns slices sorted by `mean_similarity` ascending (worst first). Recursively splits high-variance slices up to depth 3.
450
+
451
+ Returns an empty list if diffs have no tags.
452
+
453
+ ---
454
+
455
+ ### diffprompt.core.scorer
456
+
457
+ #### regression_score
458
+
459
+ ```python
460
+ def regression_score(diffs: list[DiffResult]) -> float
461
+ ```
462
+
463
+ Computes overall regression score from 0 to 100. Weighted by divergence so large behavioral changes matter more than small ones. Returns 50.0 for empty input and 100.0 if all outputs are identical.
464
+
465
+ Formula: `((weighted_sum / total_weight) + 1) / 2 * 100`
466
+ where improvements contribute `+divergence` and regressions contribute `-divergence`.
467
+
468
+ ---
469
+
470
+ #### importance_score
471
+
472
+ ```python
473
+ def importance_score(diff: DiffResult) -> float
474
+ ```
475
+
476
+ Ranks a single diff by how informative it is. Returns a float between 0 and 1.
477
+
478
+ Formula: `0.4 * divergence + 0.3 * cluster_centrality + 0.3 * surprise`
479
+ where `surprise = divergence * (1 - input_length / 50)`. Short inputs that changed a lot are surprising.
480
+
481
+ ---
482
+
483
+ #### select_key_examples
484
+
485
+ ```python
486
+ async def select_key_examples(
487
+ diffs: list[DiffResult],
488
+ local_only: bool = False,
489
+ ) -> list[KeyExample]
490
+ ```
491
+
492
+ Selects up to three key examples and generates a "why it matters" sentence for each via LLM call.
493
+
494
+ - Slot 1: Most Important (highest `importance_score`)
495
+ - Slot 2: Best Improvement (highest divergence among improvements; omitted if none)
496
+ - Slot 3: Most Surprising (high divergence on short input, excluding slot 1)
497
+
498
+ ---
499
+
500
+ ### diffprompt.models.cascade
501
+
502
+ #### call_cascade
503
+
504
+ ```python
505
+ async def call_cascade(
506
+ prompt: str,
507
+ system: str | None = None,
508
+ local_model: str = "qwen2.5:7b",
509
+ groq_model: str = "llama-3.3-70b-versatile",
510
+ local_only: bool = False,
511
+ ) -> tuple[str, str]
512
+ ```
513
+
514
+ The main LLM entry point. Tries Ollama first, falls back to Groq. Returns `(output, model_used)`.
515
+
516
+ - `prompt` — the user message
517
+ - `system` — optional system message
518
+ - `local_only` — if True, raises `RuntimeError` when Ollama is unavailable instead of falling back to Groq
519
+
520
+ Raises `RuntimeError` if all models fail.
521
+
522
+ ---
523
+
524
+ #### call_ollama
525
+
526
+ ```python
527
+ async def call_ollama(
528
+ model: str,
529
+ prompt: str,
530
+ system: str | None = None,
531
+ ) -> str | None
532
+ ```
533
+
534
+ Calls a local Ollama model. Returns `None` (never raises) if Ollama is not running or the call fails.
535
+
536
+ ---
537
+
538
+ #### call_groq
539
+
540
+ ```python
541
+ async def call_groq(
542
+ model: str,
543
+ prompt: str,
544
+ system: str | None = None,
545
+ ) -> str | None
546
+ ```
547
+
548
+ Calls the Groq API. Returns `None` if `GROQ_API_KEY` is not set or the call fails. Reads the key from `os.getenv("GROQ_API_KEY")`.