minieval-pro 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- minieval_pro/__init__.py +23 -0
- minieval_pro/cli.py +71 -0
- minieval_pro/database/__init__.py +29 -0
- minieval_pro/database/db.py +165 -0
- minieval_pro/database/update_database.py +59 -0
- minieval_pro/evaluator.py +246 -0
- minieval_pro/scorers/__init__.py +12 -0
- minieval_pro/scorers/faithfulness.py +129 -0
- minieval_pro/scorers/load_halueval.py +105 -0
- minieval_pro/scorers/relevance.py +103 -0
- minieval_pro/scorers/toxicity.py +100 -0
- minieval_pro/tests/__init__.py +0 -0
- minieval_pro/tests/test_all.py +71 -0
- minieval_pro/web/__init__.py +1 -0
- minieval_pro/web/app.py +446 -0
- minieval_pro/web/download_halueval.py +19 -0
- minieval_pro-1.0.0.dist-info/METADATA +173 -0
- minieval_pro-1.0.0.dist-info/RECORD +21 -0
- minieval_pro-1.0.0.dist-info/WHEEL +5 -0
- minieval_pro-1.0.0.dist-info/entry_points.txt +2 -0
- minieval_pro-1.0.0.dist-info/top_level.txt +1 -0
minieval_pro/__init__.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""MiniEval - A minimal evaluation framework for LLM outputs."""
|
|
2
|
+
|
|
3
|
+
# Import from the scorers folder
|
|
4
|
+
from .scorers.faithfulness import FaithfulnessScorer, FaithfulnessResult
|
|
5
|
+
from .scorers.relevance import RelevanceScorer, RelevanceResult
|
|
6
|
+
from .scorers.toxicity import ToxicityScorer, ToxicityResult
|
|
7
|
+
|
|
8
|
+
# Import from evaluator
|
|
9
|
+
from .evaluator import Evaluator, EvalResult
|
|
10
|
+
|
|
11
|
+
__version__ = "0.1.0"
|
|
12
|
+
__all__ = [
|
|
13
|
+
# Main API
|
|
14
|
+
"Evaluator",
|
|
15
|
+
"EvalResult",
|
|
16
|
+
# Individual scorers
|
|
17
|
+
"FaithfulnessScorer",
|
|
18
|
+
"FaithfulnessResult",
|
|
19
|
+
"RelevanceScorer",
|
|
20
|
+
"RelevanceResult",
|
|
21
|
+
"ToxicityScorer",
|
|
22
|
+
"ToxicityResult",
|
|
23
|
+
]
|
minieval_pro/cli.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import sys
|
|
3
|
+
import uvicorn
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def main():
|
|
8
|
+
parser = argparse.ArgumentParser(
|
|
9
|
+
prog="minieval-pro",
|
|
10
|
+
description="MiniEval Pro — LLM hallucination detection dashboard"
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
parser.add_argument(
|
|
14
|
+
"--port",
|
|
15
|
+
type=int,
|
|
16
|
+
default=8000,
|
|
17
|
+
help="Port to run dashboard on (default: 8000)"
|
|
18
|
+
)
|
|
19
|
+
parser.add_argument(
|
|
20
|
+
"--host",
|
|
21
|
+
type=str,
|
|
22
|
+
default="127.0.0.1",
|
|
23
|
+
help="Host to bind to (default: 127.0.0.1)"
|
|
24
|
+
)
|
|
25
|
+
parser.add_argument(
|
|
26
|
+
"--version",
|
|
27
|
+
action="store_true",
|
|
28
|
+
help="Show version"
|
|
29
|
+
)
|
|
30
|
+
parser.add_argument(
|
|
31
|
+
"command",
|
|
32
|
+
nargs="?",
|
|
33
|
+
choices=["init", "version"],
|
|
34
|
+
help="Command to run"
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
args = parser.parse_args()
|
|
38
|
+
|
|
39
|
+
if args.version or args.command == "version":
|
|
40
|
+
print("minieval-pro v1.0.0")
|
|
41
|
+
sys.exit(0)
|
|
42
|
+
|
|
43
|
+
if args.command == "init":
|
|
44
|
+
print("Initializing MiniEval Pro database...")
|
|
45
|
+
try:
|
|
46
|
+
from minieval_pro.web.app import init_db
|
|
47
|
+
init_db()
|
|
48
|
+
print("Database initialized.")
|
|
49
|
+
print("Run 'minieval-pro' to start the dashboard.")
|
|
50
|
+
except Exception as e:
|
|
51
|
+
print(f"Init failed: {e}")
|
|
52
|
+
sys.exit(1)
|
|
53
|
+
return
|
|
54
|
+
|
|
55
|
+
# Default — start dashboard
|
|
56
|
+
print("=" * 45)
|
|
57
|
+
print(" MiniEval Pro — LLM Hallucination Detection")
|
|
58
|
+
print("=" * 45)
|
|
59
|
+
print(f" Dashboard: http://{args.host}:{args.port}")
|
|
60
|
+
print(" Press Ctrl+C to stop")
|
|
61
|
+
print("=" * 45)
|
|
62
|
+
|
|
63
|
+
try:
|
|
64
|
+
uvicorn.run(
|
|
65
|
+
"minieval_pro.web.app:app",
|
|
66
|
+
host=args.host,
|
|
67
|
+
port=args.port,
|
|
68
|
+
reload=False
|
|
69
|
+
)
|
|
70
|
+
except KeyboardInterrupt:
|
|
71
|
+
print("\nStopped.")
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Empty file to make it a package.
|
|
2
|
+
|
|
3
|
+
__version__ = "1.0.0"
|
|
4
|
+
__author__ = "Preeti Soni"
|
|
5
|
+
|
|
6
|
+
from minieval.scorers.faithfulness import FaithfulnessScorer
|
|
7
|
+
from minieval.scorers.relevance import RelevanceScorer
|
|
8
|
+
from minieval.scorers.toxicity import ToxicityScorer
|
|
9
|
+
|
|
10
|
+
class Evaluator:
|
|
11
|
+
def __init__(self):
|
|
12
|
+
self.faithfulness = FaithfulnessScorer()
|
|
13
|
+
self.relevance = RelevanceScorer()
|
|
14
|
+
self.toxicity = ToxicityScorer()
|
|
15
|
+
|
|
16
|
+
def score(self, question: str, context: str, answer: str):
|
|
17
|
+
faithfulness = self.faithfulness.score(context, answer)
|
|
18
|
+
relevance = self.relevance.score(question, answer)
|
|
19
|
+
toxicity = self.toxicity.score(answer)
|
|
20
|
+
overall = (faithfulness + relevance) / 2
|
|
21
|
+
|
|
22
|
+
return type("Result", (), {
|
|
23
|
+
"faithfulness": faithfulness,
|
|
24
|
+
"relevance": relevance,
|
|
25
|
+
"toxicity": toxicity,
|
|
26
|
+
"overall_score": overall,
|
|
27
|
+
"passed": overall > 0.7,
|
|
28
|
+
"summary": lambda: f"Overall: {overall:.2f} | Faithfulness: {faithfulness:.2f} | Relevance: {relevance:.2f}"
|
|
29
|
+
})()
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import sqlite3
|
|
2
|
+
from datetime import datetime
|
|
3
|
+
import os
|
|
4
|
+
import uuid
|
|
5
|
+
|
|
6
|
+
# Get the project root directory
|
|
7
|
+
PROJECT_ROOT = os.path.dirname(os.path.dirname(os.path.dirname(__file__)))
|
|
8
|
+
DB_PATH = os.path.join(PROJECT_ROOT, "minieval.db")
|
|
9
|
+
|
|
10
|
+
def init_db():
|
|
11
|
+
"""Initialize the database with the evaluations table"""
|
|
12
|
+
conn = sqlite3.connect(DB_PATH)
|
|
13
|
+
cursor = conn.cursor()
|
|
14
|
+
|
|
15
|
+
cursor.execute('''
|
|
16
|
+
CREATE TABLE IF NOT EXISTS evaluations (
|
|
17
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
18
|
+
run_id TEXT UNIQUE NOT NULL,
|
|
19
|
+
timestamp TEXT NOT NULL,
|
|
20
|
+
question TEXT NOT NULL,
|
|
21
|
+
answer TEXT NOT NULL,
|
|
22
|
+
context TEXT,
|
|
23
|
+
faithfulness REAL,
|
|
24
|
+
relevance REAL,
|
|
25
|
+
toxicity REAL,
|
|
26
|
+
overall_score REAL,
|
|
27
|
+
passed BOOLEAN,
|
|
28
|
+
failure_reason TEXT,
|
|
29
|
+
model_name TEXT,
|
|
30
|
+
model_temperature REAL,
|
|
31
|
+
prompt_template TEXT,
|
|
32
|
+
evaluation_duration REAL
|
|
33
|
+
)
|
|
34
|
+
''')
|
|
35
|
+
|
|
36
|
+
conn.commit()
|
|
37
|
+
conn.close()
|
|
38
|
+
print(f"[MiniEval] Database initialized at {DB_PATH}")
|
|
39
|
+
|
|
40
|
+
def save_evaluation(
|
|
41
|
+
question, answer, context,
|
|
42
|
+
faithfulness, relevance, toxicity,
|
|
43
|
+
overall_score, passed, failure_reason=None,
|
|
44
|
+
model_name=None, model_temperature=None,
|
|
45
|
+
prompt_template=None, evaluation_duration=None
|
|
46
|
+
):
|
|
47
|
+
"""Save a single evaluation to the database"""
|
|
48
|
+
conn = sqlite3.connect(DB_PATH)
|
|
49
|
+
cursor = conn.cursor()
|
|
50
|
+
|
|
51
|
+
# Generate unique run_id
|
|
52
|
+
timestamp_str = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
53
|
+
unique_id = str(uuid.uuid4())[:8]
|
|
54
|
+
run_id = f"ev_{timestamp_str}_{unique_id}"
|
|
55
|
+
|
|
56
|
+
cursor.execute('''
|
|
57
|
+
INSERT INTO evaluations (
|
|
58
|
+
run_id, timestamp, question, answer, context,
|
|
59
|
+
faithfulness, relevance, toxicity, overall_score,
|
|
60
|
+
passed, failure_reason, model_name, model_temperature,
|
|
61
|
+
prompt_template, evaluation_duration
|
|
62
|
+
)
|
|
63
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
64
|
+
''', (
|
|
65
|
+
run_id,
|
|
66
|
+
datetime.now().isoformat(),
|
|
67
|
+
question,
|
|
68
|
+
answer,
|
|
69
|
+
context,
|
|
70
|
+
faithfulness,
|
|
71
|
+
relevance,
|
|
72
|
+
toxicity,
|
|
73
|
+
overall_score,
|
|
74
|
+
passed,
|
|
75
|
+
failure_reason,
|
|
76
|
+
model_name,
|
|
77
|
+
model_temperature,
|
|
78
|
+
prompt_template,
|
|
79
|
+
evaluation_duration
|
|
80
|
+
))
|
|
81
|
+
|
|
82
|
+
eval_id = cursor.lastrowid
|
|
83
|
+
conn.commit()
|
|
84
|
+
conn.close()
|
|
85
|
+
|
|
86
|
+
return eval_id, run_id
|
|
87
|
+
|
|
88
|
+
def get_recent_evaluations(limit=100):
|
|
89
|
+
"""Get the most recent evaluations"""
|
|
90
|
+
conn = sqlite3.connect(DB_PATH)
|
|
91
|
+
cursor = conn.cursor()
|
|
92
|
+
|
|
93
|
+
cursor.execute('''
|
|
94
|
+
SELECT id, timestamp, question, answer, overall_score, passed, failure_reason
|
|
95
|
+
FROM evaluations
|
|
96
|
+
ORDER BY timestamp DESC
|
|
97
|
+
LIMIT ?
|
|
98
|
+
''', (limit,))
|
|
99
|
+
|
|
100
|
+
rows = cursor.fetchall()
|
|
101
|
+
conn.close()
|
|
102
|
+
|
|
103
|
+
evaluations = []
|
|
104
|
+
for row in rows:
|
|
105
|
+
evaluations.append({
|
|
106
|
+
'id': row[0],
|
|
107
|
+
'timestamp': row[1],
|
|
108
|
+
'question': row[2],
|
|
109
|
+
'answer': row[3],
|
|
110
|
+
'overall_score': row[4],
|
|
111
|
+
'passed': bool(row[5]),
|
|
112
|
+
'failure_reason': row[6]
|
|
113
|
+
})
|
|
114
|
+
|
|
115
|
+
return evaluations
|
|
116
|
+
|
|
117
|
+
def get_evaluation_by_id(eval_id):
|
|
118
|
+
"""Get a single evaluation by ID"""
|
|
119
|
+
conn = sqlite3.connect(DB_PATH)
|
|
120
|
+
cursor = conn.cursor()
|
|
121
|
+
|
|
122
|
+
cursor.execute('''
|
|
123
|
+
SELECT
|
|
124
|
+
id, run_id, timestamp, question, answer, context,
|
|
125
|
+
faithfulness, relevance, toxicity, overall_score,
|
|
126
|
+
passed, failure_reason, model_name, model_temperature,
|
|
127
|
+
prompt_template, evaluation_duration
|
|
128
|
+
FROM evaluations
|
|
129
|
+
WHERE id = ?
|
|
130
|
+
''', (eval_id,))
|
|
131
|
+
|
|
132
|
+
row = cursor.fetchone()
|
|
133
|
+
conn.close()
|
|
134
|
+
|
|
135
|
+
if row:
|
|
136
|
+
return {
|
|
137
|
+
'id': row[0],
|
|
138
|
+
'run_id': row[1],
|
|
139
|
+
'timestamp': row[2],
|
|
140
|
+
'question': row[3],
|
|
141
|
+
'answer': row[4],
|
|
142
|
+
'context': row[5],
|
|
143
|
+
'faithfulness': row[6],
|
|
144
|
+
'relevance': row[7],
|
|
145
|
+
'toxicity': row[8],
|
|
146
|
+
'overall_score': row[9],
|
|
147
|
+
'passed': row[10],
|
|
148
|
+
'failure_reason': row[11],
|
|
149
|
+
'model_name': row[12],
|
|
150
|
+
'model_temperature': row[13],
|
|
151
|
+
'prompt_template': row[14],
|
|
152
|
+
'evaluation_duration': row[15]
|
|
153
|
+
}
|
|
154
|
+
return None
|
|
155
|
+
def get_correlation_metrics():
|
|
156
|
+
"""Get correlation data between MiniEval and GPT-4"""
|
|
157
|
+
import json
|
|
158
|
+
import os
|
|
159
|
+
|
|
160
|
+
metrics_file = os.path.join(os.path.dirname(os.path.dirname(os.path.dirname(__file__))), 'correlation_metrics.json')
|
|
161
|
+
|
|
162
|
+
if os.path.exists(metrics_file):
|
|
163
|
+
with open(metrics_file, 'r') as f:
|
|
164
|
+
return json.load(f)
|
|
165
|
+
return None
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import sqlite3
|
|
2
|
+
|
|
3
|
+
conn = sqlite3.connect('minieval.db')
|
|
4
|
+
cursor = conn.cursor()
|
|
5
|
+
|
|
6
|
+
# Create datasets table
|
|
7
|
+
cursor.execute('''
|
|
8
|
+
CREATE TABLE IF NOT EXISTS datasets (
|
|
9
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
10
|
+
name TEXT UNIQUE NOT NULL,
|
|
11
|
+
description TEXT,
|
|
12
|
+
filename TEXT,
|
|
13
|
+
total_samples INTEGER,
|
|
14
|
+
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
|
15
|
+
is_active INTEGER DEFAULT 0
|
|
16
|
+
)
|
|
17
|
+
''')
|
|
18
|
+
|
|
19
|
+
# Create evaluations table with dataset_id
|
|
20
|
+
cursor.execute('''
|
|
21
|
+
CREATE TABLE IF NOT EXISTS evaluation_results (
|
|
22
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
23
|
+
dataset_id INTEGER,
|
|
24
|
+
question TEXT,
|
|
25
|
+
answer TEXT,
|
|
26
|
+
ground_truth TEXT,
|
|
27
|
+
faithfulness_score REAL,
|
|
28
|
+
relevance_score REAL,
|
|
29
|
+
overall_score REAL,
|
|
30
|
+
status TEXT,
|
|
31
|
+
evaluated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
|
32
|
+
FOREIGN KEY (dataset_id) REFERENCES datasets (id)
|
|
33
|
+
)
|
|
34
|
+
''')
|
|
35
|
+
|
|
36
|
+
# Check if old evaluations table exists and migrate data
|
|
37
|
+
cursor.execute("SELECT name FROM sqlite_master WHERE type='table' AND name='evaluations'")
|
|
38
|
+
if cursor.fetchone():
|
|
39
|
+
# Create default dataset for old data
|
|
40
|
+
cursor.execute('''
|
|
41
|
+
INSERT OR IGNORE INTO datasets (name, description, is_active)
|
|
42
|
+
VALUES ('Legacy Evaluations', 'Imported from previous evaluations', 1)
|
|
43
|
+
''')
|
|
44
|
+
cursor.execute("SELECT id FROM datasets WHERE name='Legacy Evaluations'")
|
|
45
|
+
dataset_id = cursor.fetchone()[0]
|
|
46
|
+
|
|
47
|
+
# Migrate old data
|
|
48
|
+
cursor.execute('''
|
|
49
|
+
INSERT INTO evaluation_results (dataset_id, question, answer, faithfulness_score,
|
|
50
|
+
relevance_score, overall_score, status, evaluated_at)
|
|
51
|
+
SELECT ?, question, answer, faithfulness_score, relevance_score,
|
|
52
|
+
overall_score, status, evaluation_date
|
|
53
|
+
FROM evaluations
|
|
54
|
+
''', (dataset_id,))
|
|
55
|
+
print("✅ Migrated legacy data")
|
|
56
|
+
|
|
57
|
+
conn.commit()
|
|
58
|
+
conn.close()
|
|
59
|
+
print("✅ Database updated with dataset management features")
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
from dataclasses import dataclass, field
|
|
3
|
+
from typing import Optional
|
|
4
|
+
import time
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from minieval.scorers.faithfulness import FaithfulnessScorer, FaithfulnessResult
|
|
8
|
+
from minieval.scorers.relevance import RelevanceScorer, RelevanceResult
|
|
9
|
+
from minieval.scorers.toxicity import ToxicityScorer, ToxicityResult
|
|
10
|
+
|
|
11
|
+
from minieval.database.db import save_evaluation
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class EvalResult:
|
|
16
|
+
"""Complete evaluation result for one LLM output."""
|
|
17
|
+
overall: float # 0.0 to 1.0 — the single headline score
|
|
18
|
+
|
|
19
|
+
faithfulness: FaithfulnessResult
|
|
20
|
+
relevance: RelevanceResult
|
|
21
|
+
toxicity: ToxicityResult
|
|
22
|
+
|
|
23
|
+
# Weights used to calculate overall score
|
|
24
|
+
weights: dict = field(default_factory=lambda: {
|
|
25
|
+
"faithfulness": 0.5,
|
|
26
|
+
"relevance": 0.3,
|
|
27
|
+
"safety": 0.2,
|
|
28
|
+
})
|
|
29
|
+
|
|
30
|
+
def summary(self) -> str:
|
|
31
|
+
"""Human-readable summary of the evaluation."""
|
|
32
|
+
lines = [
|
|
33
|
+
f"Overall Score: {self.overall:.2f} / 1.00",
|
|
34
|
+
f"Faithfulness: {self.faithfulness.score:.2f} ({self.faithfulness.label})",
|
|
35
|
+
f"Relevance: {self.relevance.score:.2f} ({self.relevance.label})",
|
|
36
|
+
f"Toxicity: {self.toxicity.score:.2f} ({self.toxicity.label})",
|
|
37
|
+
"",
|
|
38
|
+
"Details:",
|
|
39
|
+
f" {self.faithfulness.explanation}",
|
|
40
|
+
f" {self.relevance.explanation}",
|
|
41
|
+
f" {self.toxicity.explanation}",
|
|
42
|
+
]
|
|
43
|
+
return "\n".join(lines)
|
|
44
|
+
|
|
45
|
+
def passed(self, threshold: float = 0.6) -> bool:
|
|
46
|
+
"""Quick check — did this output meet minimum quality?"""
|
|
47
|
+
return self.overall >= threshold and not self.toxicity.is_toxic
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class Evaluator:
|
|
51
|
+
"""
|
|
52
|
+
Main MiniEval API. This is what users import and use.
|
|
53
|
+
|
|
54
|
+
Usage:
|
|
55
|
+
from minieval import Evaluator
|
|
56
|
+
|
|
57
|
+
ev = Evaluator()
|
|
58
|
+
result = ev.score(
|
|
59
|
+
question = "What is the capital of France?",
|
|
60
|
+
context = "France is a country in Europe. Its capital is Paris.",
|
|
61
|
+
answer = "The capital of France is Paris.",
|
|
62
|
+
)
|
|
63
|
+
print(result.summary())
|
|
64
|
+
print(result.passed()) # True
|
|
65
|
+
"""
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
faithfulness_weight: float = 0.5,
|
|
70
|
+
relevance_weight: float = 0.3,
|
|
71
|
+
safety_weight: float = 0.2,
|
|
72
|
+
):
|
|
73
|
+
# Validate weights sum to 1.0
|
|
74
|
+
total = faithfulness_weight + relevance_weight + safety_weight
|
|
75
|
+
if abs(total - 1.0) > 0.01:
|
|
76
|
+
raise ValueError(
|
|
77
|
+
f"Weights must sum to 1.0, got {total:.2f}. "
|
|
78
|
+
f"Adjust faithfulness_weight, relevance_weight, safety_weight."
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
self.fw = faithfulness_weight
|
|
82
|
+
self.rw = relevance_weight
|
|
83
|
+
self.sw = safety_weight
|
|
84
|
+
|
|
85
|
+
# Create scorer instances — models load lazily on first use
|
|
86
|
+
self._faithfulness = FaithfulnessScorer()
|
|
87
|
+
self._relevance = RelevanceScorer()
|
|
88
|
+
self._toxicity = ToxicityScorer()
|
|
89
|
+
|
|
90
|
+
def _extract_contradiction(self, context: str, answer: str) -> str | None:
|
|
91
|
+
"""Try to find what exactly contradicts between context and answer"""
|
|
92
|
+
# Check for number differences
|
|
93
|
+
context_numbers = re.findall(r'\d+', context)
|
|
94
|
+
answer_numbers = re.findall(r'\d+', answer)
|
|
95
|
+
|
|
96
|
+
if context_numbers and answer_numbers and context_numbers[0] != answer_numbers[0]:
|
|
97
|
+
return f"Your source says {context_numbers[0]}, but the answer says {answer_numbers[0]}."
|
|
98
|
+
|
|
99
|
+
# Check for percentage differences
|
|
100
|
+
context_percent = re.findall(r'(\d+)%', context)
|
|
101
|
+
answer_percent = re.findall(r'(\d+)%', answer)
|
|
102
|
+
|
|
103
|
+
if context_percent and answer_percent and context_percent[0] != answer_percent[0]:
|
|
104
|
+
return f"Your source says {context_percent[0]}%, but the answer says {answer_percent[0]}%."
|
|
105
|
+
|
|
106
|
+
return None
|
|
107
|
+
|
|
108
|
+
def score(
|
|
109
|
+
self,
|
|
110
|
+
question: str, # what the user asked
|
|
111
|
+
context: str, # what information the LLM was given
|
|
112
|
+
answer: str, # what the LLM responded
|
|
113
|
+
) -> EvalResult:
|
|
114
|
+
"""
|
|
115
|
+
Evaluate one LLM output completely.
|
|
116
|
+
Runs all three scorers and returns a combined result.
|
|
117
|
+
"""
|
|
118
|
+
# Start timing
|
|
119
|
+
start_time = time.time()
|
|
120
|
+
|
|
121
|
+
# Run all three scorers
|
|
122
|
+
faith_result = self._faithfulness.score(context, answer)
|
|
123
|
+
rel_result = self._relevance.score(question, answer)
|
|
124
|
+
tox_result = self._toxicity.score(answer)
|
|
125
|
+
|
|
126
|
+
# safety_score: 1.0 = completely safe, 0.0 = completely toxic
|
|
127
|
+
# We invert toxicity because high toxicity = bad quality
|
|
128
|
+
safety_score = 1.0 - tox_result.score
|
|
129
|
+
|
|
130
|
+
# Weighted combination
|
|
131
|
+
overall = (
|
|
132
|
+
(faith_result.score * self.fw) +
|
|
133
|
+
(rel_result.score * self.rw) +
|
|
134
|
+
(safety_score * self.sw)
|
|
135
|
+
)
|
|
136
|
+
overall = round(max(0.0, min(1.0, overall)), 4)
|
|
137
|
+
|
|
138
|
+
# Calculate duration
|
|
139
|
+
evaluation_duration = round(time.time() - start_time, 3)
|
|
140
|
+
|
|
141
|
+
# Create EvalResult first
|
|
142
|
+
result = EvalResult(
|
|
143
|
+
overall=overall,
|
|
144
|
+
faithfulness=faith_result,
|
|
145
|
+
relevance=rel_result,
|
|
146
|
+
toxicity=tox_result,
|
|
147
|
+
weights={
|
|
148
|
+
"faithfulness": self.fw,
|
|
149
|
+
"relevance": self.rw,
|
|
150
|
+
"safety": self.sw,
|
|
151
|
+
},
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
# Save to database
|
|
155
|
+
threshold = 0.6
|
|
156
|
+
passed = result.passed(threshold)
|
|
157
|
+
|
|
158
|
+
# ========== PLAIN ENGLISH EXPLAINER ==========
|
|
159
|
+
failure_reason = None
|
|
160
|
+
if not passed:
|
|
161
|
+
reasons = []
|
|
162
|
+
|
|
163
|
+
# Check for contradiction
|
|
164
|
+
contradiction = self._extract_contradiction(context, answer)
|
|
165
|
+
|
|
166
|
+
# Faithfulness / Hallucination check
|
|
167
|
+
if faith_result.score < 0.5:
|
|
168
|
+
if contradiction:
|
|
169
|
+
reasons.append(f"❌ HALLUCINATION: {contradiction}")
|
|
170
|
+
else:
|
|
171
|
+
# Try to show what the answer said vs context
|
|
172
|
+
context_preview = context[:150] if context else "your source document"
|
|
173
|
+
answer_preview = answer[:150]
|
|
174
|
+
reasons.append(f"❌ HALLUCINATION: The answer says \"{answer_preview}...\"")
|
|
175
|
+
reasons.append(f" But this information is NOT found in {context_preview}...")
|
|
176
|
+
reasons.append(f" The LLM invented information that wasn't there. Confidence: {faith_result.score:.0%}")
|
|
177
|
+
|
|
178
|
+
# Relevance check
|
|
179
|
+
if rel_result.score < 0.5:
|
|
180
|
+
reasons.append(f"❌ IRRELEVANT ANSWER: The answer doesn't address the question.")
|
|
181
|
+
reasons.append(f" Question: \"{question[:100]}...\"")
|
|
182
|
+
reasons.append(f" Answer talked about: \"{answer[:100]}...\"")
|
|
183
|
+
reasons.append(f" Relevance score: {rel_result.score:.0%} (needs >50%)")
|
|
184
|
+
|
|
185
|
+
# Toxicity check
|
|
186
|
+
if tox_result.is_toxic:
|
|
187
|
+
reasons.append(f"❌ TOXIC CONTENT: The answer contains unsafe or inappropriate language.")
|
|
188
|
+
reasons.append(f" Offending text: \"{answer[:150]}...\"")
|
|
189
|
+
reasons.append(f" Toxicity score: {tox_result.score:.0%} (should be <30%)")
|
|
190
|
+
|
|
191
|
+
# Low overall score (but no specific issue found)
|
|
192
|
+
if overall < threshold and faith_result.score >= 0.5 and rel_result.score >= 0.5 and not tox_result.is_toxic:
|
|
193
|
+
reasons.append(f"⚠️ LOW QUALITY SCORE: {overall:.0%} (needs >{threshold:.0%})")
|
|
194
|
+
reasons.append(f" The answer is faithful and relevant, but quality is below standard.")
|
|
195
|
+
reasons.append(f" Consider improving answer clarity, adding more details, or making it more specific.")
|
|
196
|
+
|
|
197
|
+
failure_reason = "\n\n".join(reasons)
|
|
198
|
+
|
|
199
|
+
if not failure_reason and passed:
|
|
200
|
+
failure_reason = f"✅ PASSED: Quality score {overall:.0%} meets threshold (>70%). Answer is faithful, relevant, and safe."
|
|
201
|
+
|
|
202
|
+
# Hyperparameters
|
|
203
|
+
model_name = "minieval-v1"
|
|
204
|
+
model_temperature = 0.7
|
|
205
|
+
prompt_template = "default"
|
|
206
|
+
|
|
207
|
+
# Save to database with hyperparameters
|
|
208
|
+
save_evaluation(
|
|
209
|
+
question=question,
|
|
210
|
+
answer=answer,
|
|
211
|
+
context=context,
|
|
212
|
+
faithfulness=faith_result.score,
|
|
213
|
+
relevance=rel_result.score,
|
|
214
|
+
toxicity=tox_result.score,
|
|
215
|
+
overall_score=overall,
|
|
216
|
+
passed=passed,
|
|
217
|
+
failure_reason=failure_reason,
|
|
218
|
+
model_name=model_name,
|
|
219
|
+
model_temperature=model_temperature,
|
|
220
|
+
prompt_template=prompt_template,
|
|
221
|
+
evaluation_duration=evaluation_duration
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
return result
|
|
225
|
+
|
|
226
|
+
def score_batch(
|
|
227
|
+
self,
|
|
228
|
+
items: list[dict], # list of {question, context, answer}
|
|
229
|
+
) -> list[EvalResult]:
|
|
230
|
+
"""
|
|
231
|
+
Evaluate multiple outputs at once.
|
|
232
|
+
|
|
233
|
+
Usage:
|
|
234
|
+
results = ev.score_batch([
|
|
235
|
+
{"question": "...", "context": "...", "answer": "..."},
|
|
236
|
+
{"question": "...", "context": "...", "answer": "..."},
|
|
237
|
+
])
|
|
238
|
+
"""
|
|
239
|
+
return [
|
|
240
|
+
self.score(
|
|
241
|
+
question=item["question"],
|
|
242
|
+
context=item["context"],
|
|
243
|
+
answer=item["answer"],
|
|
244
|
+
)
|
|
245
|
+
for item in items
|
|
246
|
+
]
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from .faithfulness import FaithfulnessScorer, FaithfulnessResult
|
|
2
|
+
from .relevance import RelevanceScorer, RelevanceResult
|
|
3
|
+
from .toxicity import ToxicityScorer, ToxicityResult
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"FaithfulnessScorer",
|
|
7
|
+
"FaithfulnessResult",
|
|
8
|
+
"RelevanceScorer",
|
|
9
|
+
"RelevanceResult",
|
|
10
|
+
"ToxicityScorer",
|
|
11
|
+
"ToxicityResult",
|
|
12
|
+
]
|