minieval-pro 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,23 @@
1
+ """MiniEval - A minimal evaluation framework for LLM outputs."""
2
+
3
+ # Import from the scorers folder
4
+ from .scorers.faithfulness import FaithfulnessScorer, FaithfulnessResult
5
+ from .scorers.relevance import RelevanceScorer, RelevanceResult
6
+ from .scorers.toxicity import ToxicityScorer, ToxicityResult
7
+
8
+ # Import from evaluator
9
+ from .evaluator import Evaluator, EvalResult
10
+
11
+ __version__ = "0.1.0"
12
+ __all__ = [
13
+ # Main API
14
+ "Evaluator",
15
+ "EvalResult",
16
+ # Individual scorers
17
+ "FaithfulnessScorer",
18
+ "FaithfulnessResult",
19
+ "RelevanceScorer",
20
+ "RelevanceResult",
21
+ "ToxicityScorer",
22
+ "ToxicityResult",
23
+ ]
minieval_pro/cli.py ADDED
@@ -0,0 +1,71 @@
1
+ import argparse
2
+ import sys
3
+ import uvicorn
4
+ from pathlib import Path
5
+
6
+
7
+ def main():
8
+ parser = argparse.ArgumentParser(
9
+ prog="minieval-pro",
10
+ description="MiniEval Pro — LLM hallucination detection dashboard"
11
+ )
12
+
13
+ parser.add_argument(
14
+ "--port",
15
+ type=int,
16
+ default=8000,
17
+ help="Port to run dashboard on (default: 8000)"
18
+ )
19
+ parser.add_argument(
20
+ "--host",
21
+ type=str,
22
+ default="127.0.0.1",
23
+ help="Host to bind to (default: 127.0.0.1)"
24
+ )
25
+ parser.add_argument(
26
+ "--version",
27
+ action="store_true",
28
+ help="Show version"
29
+ )
30
+ parser.add_argument(
31
+ "command",
32
+ nargs="?",
33
+ choices=["init", "version"],
34
+ help="Command to run"
35
+ )
36
+
37
+ args = parser.parse_args()
38
+
39
+ if args.version or args.command == "version":
40
+ print("minieval-pro v1.0.0")
41
+ sys.exit(0)
42
+
43
+ if args.command == "init":
44
+ print("Initializing MiniEval Pro database...")
45
+ try:
46
+ from minieval_pro.web.app import init_db
47
+ init_db()
48
+ print("Database initialized.")
49
+ print("Run 'minieval-pro' to start the dashboard.")
50
+ except Exception as e:
51
+ print(f"Init failed: {e}")
52
+ sys.exit(1)
53
+ return
54
+
55
+ # Default — start dashboard
56
+ print("=" * 45)
57
+ print(" MiniEval Pro — LLM Hallucination Detection")
58
+ print("=" * 45)
59
+ print(f" Dashboard: http://{args.host}:{args.port}")
60
+ print(" Press Ctrl+C to stop")
61
+ print("=" * 45)
62
+
63
+ try:
64
+ uvicorn.run(
65
+ "minieval_pro.web.app:app",
66
+ host=args.host,
67
+ port=args.port,
68
+ reload=False
69
+ )
70
+ except KeyboardInterrupt:
71
+ print("\nStopped.")
@@ -0,0 +1,29 @@
1
+ # Empty file to make it a package.
2
+
3
+ __version__ = "1.0.0"
4
+ __author__ = "Preeti Soni"
5
+
6
+ from minieval.scorers.faithfulness import FaithfulnessScorer
7
+ from minieval.scorers.relevance import RelevanceScorer
8
+ from minieval.scorers.toxicity import ToxicityScorer
9
+
10
+ class Evaluator:
11
+ def __init__(self):
12
+ self.faithfulness = FaithfulnessScorer()
13
+ self.relevance = RelevanceScorer()
14
+ self.toxicity = ToxicityScorer()
15
+
16
+ def score(self, question: str, context: str, answer: str):
17
+ faithfulness = self.faithfulness.score(context, answer)
18
+ relevance = self.relevance.score(question, answer)
19
+ toxicity = self.toxicity.score(answer)
20
+ overall = (faithfulness + relevance) / 2
21
+
22
+ return type("Result", (), {
23
+ "faithfulness": faithfulness,
24
+ "relevance": relevance,
25
+ "toxicity": toxicity,
26
+ "overall_score": overall,
27
+ "passed": overall > 0.7,
28
+ "summary": lambda: f"Overall: {overall:.2f} | Faithfulness: {faithfulness:.2f} | Relevance: {relevance:.2f}"
29
+ })()
@@ -0,0 +1,165 @@
1
+ import sqlite3
2
+ from datetime import datetime
3
+ import os
4
+ import uuid
5
+
6
+ # Get the project root directory
7
+ PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(__file__)))
8
+ DB_PATH = os.path.join(PROJECT_ROOT, "minieval.db")
9
+
10
+ def init_db():
11
+ """Initialize the database with the evaluations table"""
12
+ conn = sqlite3.connect(DB_PATH)
13
+ cursor = conn.cursor()
14
+
15
+ cursor.execute('''
16
+ CREATE TABLE IF NOT EXISTS evaluations (
17
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
18
+ run_id TEXT UNIQUE NOT NULL,
19
+ timestamp TEXT NOT NULL,
20
+ question TEXT NOT NULL,
21
+ answer TEXT NOT NULL,
22
+ context TEXT,
23
+ faithfulness REAL,
24
+ relevance REAL,
25
+ toxicity REAL,
26
+ overall_score REAL,
27
+ passed BOOLEAN,
28
+ failure_reason TEXT,
29
+ model_name TEXT,
30
+ model_temperature REAL,
31
+ prompt_template TEXT,
32
+ evaluation_duration REAL
33
+ )
34
+ ''')
35
+
36
+ conn.commit()
37
+ conn.close()
38
+ print(f"[MiniEval] Database initialized at {DB_PATH}")
39
+
40
+ def save_evaluation(
41
+ question, answer, context,
42
+ faithfulness, relevance, toxicity,
43
+ overall_score, passed, failure_reason=None,
44
+ model_name=None, model_temperature=None,
45
+ prompt_template=None, evaluation_duration=None
46
+ ):
47
+ """Save a single evaluation to the database"""
48
+ conn = sqlite3.connect(DB_PATH)
49
+ cursor = conn.cursor()
50
+
51
+ # Generate unique run_id
52
+ timestamp_str = datetime.now().strftime("%Y%m%d_%H%M%S")
53
+ unique_id = str(uuid.uuid4())[:8]
54
+ run_id = f"ev_{timestamp_str}_{unique_id}"
55
+
56
+ cursor.execute('''
57
+ INSERT INTO evaluations (
58
+ run_id, timestamp, question, answer, context,
59
+ faithfulness, relevance, toxicity, overall_score,
60
+ passed, failure_reason, model_name, model_temperature,
61
+ prompt_template, evaluation_duration
62
+ )
63
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
64
+ ''', (
65
+ run_id,
66
+ datetime.now().isoformat(),
67
+ question,
68
+ answer,
69
+ context,
70
+ faithfulness,
71
+ relevance,
72
+ toxicity,
73
+ overall_score,
74
+ passed,
75
+ failure_reason,
76
+ model_name,
77
+ model_temperature,
78
+ prompt_template,
79
+ evaluation_duration
80
+ ))
81
+
82
+ eval_id = cursor.lastrowid
83
+ conn.commit()
84
+ conn.close()
85
+
86
+ return eval_id, run_id
87
+
88
+ def get_recent_evaluations(limit=100):
89
+ """Get the most recent evaluations"""
90
+ conn = sqlite3.connect(DB_PATH)
91
+ cursor = conn.cursor()
92
+
93
+ cursor.execute('''
94
+ SELECT id, timestamp, question, answer, overall_score, passed, failure_reason
95
+ FROM evaluations
96
+ ORDER BY timestamp DESC
97
+ LIMIT ?
98
+ ''', (limit,))
99
+
100
+ rows = cursor.fetchall()
101
+ conn.close()
102
+
103
+ evaluations = []
104
+ for row in rows:
105
+ evaluations.append({
106
+ 'id': row[0],
107
+ 'timestamp': row[1],
108
+ 'question': row[2],
109
+ 'answer': row[3],
110
+ 'overall_score': row[4],
111
+ 'passed': bool(row[5]),
112
+ 'failure_reason': row[6]
113
+ })
114
+
115
+ return evaluations
116
+
117
+ def get_evaluation_by_id(eval_id):
118
+ """Get a single evaluation by ID"""
119
+ conn = sqlite3.connect(DB_PATH)
120
+ cursor = conn.cursor()
121
+
122
+ cursor.execute('''
123
+ SELECT
124
+ id, run_id, timestamp, question, answer, context,
125
+ faithfulness, relevance, toxicity, overall_score,
126
+ passed, failure_reason, model_name, model_temperature,
127
+ prompt_template, evaluation_duration
128
+ FROM evaluations
129
+ WHERE id = ?
130
+ ''', (eval_id,))
131
+
132
+ row = cursor.fetchone()
133
+ conn.close()
134
+
135
+ if row:
136
+ return {
137
+ 'id': row[0],
138
+ 'run_id': row[1],
139
+ 'timestamp': row[2],
140
+ 'question': row[3],
141
+ 'answer': row[4],
142
+ 'context': row[5],
143
+ 'faithfulness': row[6],
144
+ 'relevance': row[7],
145
+ 'toxicity': row[8],
146
+ 'overall_score': row[9],
147
+ 'passed': row[10],
148
+ 'failure_reason': row[11],
149
+ 'model_name': row[12],
150
+ 'model_temperature': row[13],
151
+ 'prompt_template': row[14],
152
+ 'evaluation_duration': row[15]
153
+ }
154
+ return None
155
+ def get_correlation_metrics():
156
+ """Get correlation data between MiniEval and GPT-4"""
157
+ import json
158
+ import os
159
+
160
+ metrics_file = os.path.join(os.path.dirname(os.path.dirname(os.path.dirname(__file__))), 'correlation_metrics.json')
161
+
162
+ if os.path.exists(metrics_file):
163
+ with open(metrics_file, 'r') as f:
164
+ return json.load(f)
165
+ return None
@@ -0,0 +1,59 @@
1
+ import sqlite3
2
+
3
+ conn = sqlite3.connect('minieval.db')
4
+ cursor = conn.cursor()
5
+
6
+ # Create datasets table
7
+ cursor.execute('''
8
+ CREATE TABLE IF NOT EXISTS datasets (
9
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
10
+ name TEXT UNIQUE NOT NULL,
11
+ description TEXT,
12
+ filename TEXT,
13
+ total_samples INTEGER,
14
+ created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
15
+ is_active INTEGER DEFAULT 0
16
+ )
17
+ ''')
18
+
19
+ # Create evaluations table with dataset_id
20
+ cursor.execute('''
21
+ CREATE TABLE IF NOT EXISTS evaluation_results (
22
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
23
+ dataset_id INTEGER,
24
+ question TEXT,
25
+ answer TEXT,
26
+ ground_truth TEXT,
27
+ faithfulness_score REAL,
28
+ relevance_score REAL,
29
+ overall_score REAL,
30
+ status TEXT,
31
+ evaluated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
32
+ FOREIGN KEY (dataset_id) REFERENCES datasets (id)
33
+ )
34
+ ''')
35
+
36
+ # Check if old evaluations table exists and migrate data
37
+ cursor.execute("SELECT name FROM sqlite_master WHERE type='table' AND name='evaluations'")
38
+ if cursor.fetchone():
39
+ # Create default dataset for old data
40
+ cursor.execute('''
41
+ INSERT OR IGNORE INTO datasets (name, description, is_active)
42
+ VALUES ('Legacy Evaluations', 'Imported from previous evaluations', 1)
43
+ ''')
44
+ cursor.execute("SELECT id FROM datasets WHERE name='Legacy Evaluations'")
45
+ dataset_id = cursor.fetchone()[0]
46
+
47
+ # Migrate old data
48
+ cursor.execute('''
49
+ INSERT INTO evaluation_results (dataset_id, question, answer, faithfulness_score,
50
+ relevance_score, overall_score, status, evaluated_at)
51
+ SELECT ?, question, answer, faithfulness_score, relevance_score,
52
+ overall_score, status, evaluation_date
53
+ FROM evaluations
54
+ ''', (dataset_id,))
55
+ print("✅ Migrated legacy data")
56
+
57
+ conn.commit()
58
+ conn.close()
59
+ print("✅ Database updated with dataset management features")
@@ -0,0 +1,246 @@
1
+ from __future__ import annotations
2
+ from dataclasses import dataclass, field
3
+ from typing import Optional
4
+ import time
5
+ import re
6
+
7
+ from minieval.scorers.faithfulness import FaithfulnessScorer, FaithfulnessResult
8
+ from minieval.scorers.relevance import RelevanceScorer, RelevanceResult
9
+ from minieval.scorers.toxicity import ToxicityScorer, ToxicityResult
10
+
11
+ from minieval.database.db import save_evaluation
12
+
13
+
14
+ @dataclass
15
+ class EvalResult:
16
+ """Complete evaluation result for one LLM output."""
17
+ overall: float # 0.0 to 1.0 — the single headline score
18
+
19
+ faithfulness: FaithfulnessResult
20
+ relevance: RelevanceResult
21
+ toxicity: ToxicityResult
22
+
23
+ # Weights used to calculate overall score
24
+ weights: dict = field(default_factory=lambda: {
25
+ "faithfulness": 0.5,
26
+ "relevance": 0.3,
27
+ "safety": 0.2,
28
+ })
29
+
30
+ def summary(self) -> str:
31
+ """Human-readable summary of the evaluation."""
32
+ lines = [
33
+ f"Overall Score: {self.overall:.2f} / 1.00",
34
+ f"Faithfulness: {self.faithfulness.score:.2f} ({self.faithfulness.label})",
35
+ f"Relevance: {self.relevance.score:.2f} ({self.relevance.label})",
36
+ f"Toxicity: {self.toxicity.score:.2f} ({self.toxicity.label})",
37
+ "",
38
+ "Details:",
39
+ f" {self.faithfulness.explanation}",
40
+ f" {self.relevance.explanation}",
41
+ f" {self.toxicity.explanation}",
42
+ ]
43
+ return "\n".join(lines)
44
+
45
+ def passed(self, threshold: float = 0.6) -> bool:
46
+ """Quick check — did this output meet minimum quality?"""
47
+ return self.overall >= threshold and not self.toxicity.is_toxic
48
+
49
+
50
+ class Evaluator:
51
+ """
52
+ Main MiniEval API. This is what users import and use.
53
+
54
+ Usage:
55
+ from minieval import Evaluator
56
+
57
+ ev = Evaluator()
58
+ result = ev.score(
59
+ question = "What is the capital of France?",
60
+ context = "France is a country in Europe. Its capital is Paris.",
61
+ answer = "The capital of France is Paris.",
62
+ )
63
+ print(result.summary())
64
+ print(result.passed()) # True
65
+ """
66
+
67
+ def __init__(
68
+ self,
69
+ faithfulness_weight: float = 0.5,
70
+ relevance_weight: float = 0.3,
71
+ safety_weight: float = 0.2,
72
+ ):
73
+ # Validate weights sum to 1.0
74
+ total = faithfulness_weight + relevance_weight + safety_weight
75
+ if abs(total - 1.0) > 0.01:
76
+ raise ValueError(
77
+ f"Weights must sum to 1.0, got {total:.2f}. "
78
+ f"Adjust faithfulness_weight, relevance_weight, safety_weight."
79
+ )
80
+
81
+ self.fw = faithfulness_weight
82
+ self.rw = relevance_weight
83
+ self.sw = safety_weight
84
+
85
+ # Create scorer instances — models load lazily on first use
86
+ self._faithfulness = FaithfulnessScorer()
87
+ self._relevance = RelevanceScorer()
88
+ self._toxicity = ToxicityScorer()
89
+
90
+ def _extract_contradiction(self, context: str, answer: str) -> str | None:
91
+ """Try to find what exactly contradicts between context and answer"""
92
+ # Check for number differences
93
+ context_numbers = re.findall(r'\d+', context)
94
+ answer_numbers = re.findall(r'\d+', answer)
95
+
96
+ if context_numbers and answer_numbers and context_numbers[0] != answer_numbers[0]:
97
+ return f"Your source says {context_numbers[0]}, but the answer says {answer_numbers[0]}."
98
+
99
+ # Check for percentage differences
100
+ context_percent = re.findall(r'(\d+)%', context)
101
+ answer_percent = re.findall(r'(\d+)%', answer)
102
+
103
+ if context_percent and answer_percent and context_percent[0] != answer_percent[0]:
104
+ return f"Your source says {context_percent[0]}%, but the answer says {answer_percent[0]}%."
105
+
106
+ return None
107
+
108
+ def score(
109
+ self,
110
+ question: str, # what the user asked
111
+ context: str, # what information the LLM was given
112
+ answer: str, # what the LLM responded
113
+ ) -> EvalResult:
114
+ """
115
+ Evaluate one LLM output completely.
116
+ Runs all three scorers and returns a combined result.
117
+ """
118
+ # Start timing
119
+ start_time = time.time()
120
+
121
+ # Run all three scorers
122
+ faith_result = self._faithfulness.score(context, answer)
123
+ rel_result = self._relevance.score(question, answer)
124
+ tox_result = self._toxicity.score(answer)
125
+
126
+ # safety_score: 1.0 = completely safe, 0.0 = completely toxic
127
+ # We invert toxicity because high toxicity = bad quality
128
+ safety_score = 1.0 - tox_result.score
129
+
130
+ # Weighted combination
131
+ overall = (
132
+ (faith_result.score * self.fw) +
133
+ (rel_result.score * self.rw) +
134
+ (safety_score * self.sw)
135
+ )
136
+ overall = round(max(0.0, min(1.0, overall)), 4)
137
+
138
+ # Calculate duration
139
+ evaluation_duration = round(time.time() - start_time, 3)
140
+
141
+ # Create EvalResult first
142
+ result = EvalResult(
143
+ overall=overall,
144
+ faithfulness=faith_result,
145
+ relevance=rel_result,
146
+ toxicity=tox_result,
147
+ weights={
148
+ "faithfulness": self.fw,
149
+ "relevance": self.rw,
150
+ "safety": self.sw,
151
+ },
152
+ )
153
+
154
+ # Save to database
155
+ threshold = 0.6
156
+ passed = result.passed(threshold)
157
+
158
+ # ========== PLAIN ENGLISH EXPLAINER ==========
159
+ failure_reason = None
160
+ if not passed:
161
+ reasons = []
162
+
163
+ # Check for contradiction
164
+ contradiction = self._extract_contradiction(context, answer)
165
+
166
+ # Faithfulness / Hallucination check
167
+ if faith_result.score < 0.5:
168
+ if contradiction:
169
+ reasons.append(f"❌ HALLUCINATION: {contradiction}")
170
+ else:
171
+ # Try to show what the answer said vs context
172
+ context_preview = context[:150] if context else "your source document"
173
+ answer_preview = answer[:150]
174
+ reasons.append(f"❌ HALLUCINATION: The answer says \"{answer_preview}...\"")
175
+ reasons.append(f" But this information is NOT found in {context_preview}...")
176
+ reasons.append(f" The LLM invented information that wasn't there. Confidence: {faith_result.score:.0%}")
177
+
178
+ # Relevance check
179
+ if rel_result.score < 0.5:
180
+ reasons.append(f"❌ IRRELEVANT ANSWER: The answer doesn't address the question.")
181
+ reasons.append(f" Question: \"{question[:100]}...\"")
182
+ reasons.append(f" Answer talked about: \"{answer[:100]}...\"")
183
+ reasons.append(f" Relevance score: {rel_result.score:.0%} (needs >50%)")
184
+
185
+ # Toxicity check
186
+ if tox_result.is_toxic:
187
+ reasons.append(f"❌ TOXIC CONTENT: The answer contains unsafe or inappropriate language.")
188
+ reasons.append(f" Offending text: \"{answer[:150]}...\"")
189
+ reasons.append(f" Toxicity score: {tox_result.score:.0%} (should be <30%)")
190
+
191
+ # Low overall score (but no specific issue found)
192
+ if overall < threshold and faith_result.score >= 0.5 and rel_result.score >= 0.5 and not tox_result.is_toxic:
193
+ reasons.append(f"⚠️ LOW QUALITY SCORE: {overall:.0%} (needs >{threshold:.0%})")
194
+ reasons.append(f" The answer is faithful and relevant, but quality is below standard.")
195
+ reasons.append(f" Consider improving answer clarity, adding more details, or making it more specific.")
196
+
197
+ failure_reason = "\n\n".join(reasons)
198
+
199
+ if not failure_reason and passed:
200
+ failure_reason = f"✅ PASSED: Quality score {overall:.0%} meets threshold (>70%). Answer is faithful, relevant, and safe."
201
+
202
+ # Hyperparameters
203
+ model_name = "minieval-v1"
204
+ model_temperature = 0.7
205
+ prompt_template = "default"
206
+
207
+ # Save to database with hyperparameters
208
+ save_evaluation(
209
+ question=question,
210
+ answer=answer,
211
+ context=context,
212
+ faithfulness=faith_result.score,
213
+ relevance=rel_result.score,
214
+ toxicity=tox_result.score,
215
+ overall_score=overall,
216
+ passed=passed,
217
+ failure_reason=failure_reason,
218
+ model_name=model_name,
219
+ model_temperature=model_temperature,
220
+ prompt_template=prompt_template,
221
+ evaluation_duration=evaluation_duration
222
+ )
223
+
224
+ return result
225
+
226
+ def score_batch(
227
+ self,
228
+ items: list[dict], # list of {question, context, answer}
229
+ ) -> list[EvalResult]:
230
+ """
231
+ Evaluate multiple outputs at once.
232
+
233
+ Usage:
234
+ results = ev.score_batch([
235
+ {"question": "...", "context": "...", "answer": "..."},
236
+ {"question": "...", "context": "...", "answer": "..."},
237
+ ])
238
+ """
239
+ return [
240
+ self.score(
241
+ question=item["question"],
242
+ context=item["context"],
243
+ answer=item["answer"],
244
+ )
245
+ for item in items
246
+ ]
@@ -0,0 +1,12 @@
1
+ from .faithfulness import FaithfulnessScorer, FaithfulnessResult
2
+ from .relevance import RelevanceScorer, RelevanceResult
3
+ from .toxicity import ToxicityScorer, ToxicityResult
4
+
5
+ __all__ = [
6
+ "FaithfulnessScorer",
7
+ "FaithfulnessResult",
8
+ "RelevanceScorer",
9
+ "RelevanceResult",
10
+ "ToxicityScorer",
11
+ "ToxicityResult",
12
+ ]