clef-evals 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
clef_evals/__init__.py ADDED
@@ -0,0 +1,117 @@
1
+ """clef-evals: Calibration-first evaluation toolkit for Cloudflare's Clef."""
2
+
3
+ import os
4
+ from dataclasses import dataclass
5
+ from typing import Optional
6
+
7
+ import httpx
8
+ import numpy as np
9
+
10
+
11
+ def ece(confidences: list[float], correct: list[bool], n_bins: int = 15) -> float:
12
+ """Expected Calibration Error."""
13
+ confs = np.array(confidences)
14
+ corr = np.array([1.0 if c else 0.0 for c in correct])
15
+ bins = np.linspace(0, 1, n_bins + 1)
16
+ ece_val = 0.0
17
+ for i in range(n_bins):
18
+ mask = (confs > bins[i]) & (confs <= bins[i + 1])
19
+ if mask.sum() == 0:
20
+ continue
21
+ bin_conf = confs[mask].mean()
22
+ bin_acc = corr[mask].mean()
23
+ ece_val += (mask.sum() / len(confs)) * abs(bin_acc - bin_conf)
24
+ return float(ece_val)
25
+
26
+
27
+ def brier_score(confidences: list[float], correct: list[bool]) -> float:
28
+ """Brier score: mean squared error of probabilistic predictions."""
29
+ confs = np.array(confidences)
30
+ corr = np.array([1.0 if c else 0.0 for c in correct])
31
+ return float(np.mean((confs - corr) ** 2))
32
+
33
+
34
+ @dataclass
35
+ class ClefEvalResult:
36
+ accuracy: float
37
+ ece: float
38
+ brier: float
39
+ n_samples: int
40
+ correct: list[bool]
41
+ confidences: list[float]
42
+
43
+
44
+ @dataclass
45
+ class ClefJudge:
46
+ """Run Clef as an LLM-as-judge over an eval set."""
47
+
48
+ account_id: str = os.environ.get("CLEF_ACCOUNT_ID", "")
49
+ api_token: str = os.environ.get("CLEF_API_TOKEN", "")
50
+ model: str = "@cf/cloudflare/clef-flash"
51
+ base_url: str = "https://api.cloudflare.com/client/v4"
52
+ timeout: float = 60.0
53
+
54
+ def __post_init__(self):
55
+ if not self.account_id:
56
+ raise ValueError("Set CLEF_ACCOUNT_ID")
57
+
58
+ def judge_choice(self, state: str, question: str, choices: list[str]) -> tuple[str, float]:
59
+ """Ask Clef to choose one option. Returns (choice, confidence)."""
60
+ payload = {
61
+ "state": state,
62
+ "questions": {
63
+ question: {
64
+ "type": "choice",
65
+ "choices": choices,
66
+ },
67
+ "confidence": {
68
+ "type": "score",
69
+ "context": "How confident are you?",
70
+ "levels": ["very_low", "low", "medium", "high", "very_high"],
71
+ },
72
+ },
73
+ }
74
+ resp = httpx.post(
75
+ f"{self.base_url}/accounts/{self.account_id}/ai/run/{self.model}",
76
+ json=payload,
77
+ headers={"Authorization": f"Bearer {self.api_token}"},
78
+ timeout=self.timeout,
79
+ )
80
+ resp.raise_for_status()
81
+ result = resp.json().get("result", {})
82
+ choice = result.get(question, choices[-1])
83
+ conf_map = {"very_low": 0.1, "low": 0.3, "medium": 0.5, "high": 0.7, "very_high": 0.9}
84
+ confidence = conf_map.get(result.get("confidence", "low"), 0.3)
85
+ return choice, confidence
86
+
87
+ def evaluate(
88
+ self,
89
+ eval_set: list[dict],
90
+ state_key: str = "state",
91
+ question_key: str = "question",
92
+ choices_key: str = "choices",
93
+ gold_key: str = "gold",
94
+ ) -> ClefEvalResult:
95
+ """Run judge over an eval set and compute calibration metrics."""
96
+ correct_flags, confidence_list = [], []
97
+ for item in eval_set:
98
+ pred, conf = self.judge_choice(
99
+ item[state_key], item[question_key], item[choices_key]
100
+ )
101
+ is_correct = pred == item[gold_key]
102
+ correct_flags.append(is_correct)
103
+ confidence_list.append(conf)
104
+
105
+ accuracy = sum(correct_flags) / len(correct_flags) if correct_flags else 0.0
106
+ return ClefEvalResult(
107
+ accuracy=accuracy,
108
+ ece=ece(confidence_list, correct_flags),
109
+ brier=brier_score(confidence_list, correct_flags),
110
+ n_samples=len(correct_flags),
111
+ correct=correct_flags,
112
+ confidences=confidence_list,
113
+ )
114
+
115
+
116
+ __version__ = "0.1.0"
117
+ __all__ = ["ClefJudge", "ClefEvalResult", "ece", "brier_score"]
clef_evals/cli.py ADDED
@@ -0,0 +1,16 @@
1
+ """CLI for clef-evals."""
2
+ import argparse
3
+ import json
4
+
5
+ from . import ClefJudge
6
+
7
+
8
+ def main():
9
+ p = argparse.ArgumentParser(prog="clef-eval", description="Evaluate Clef as judge")
10
+ p.add_argument("--json", action="store_true")
11
+ args = p.parse_args()
12
+ print("See README for usage examples.")
13
+
14
+
15
+ if __name__ == "__main__":
16
+ main()
@@ -0,0 +1,49 @@
1
+ Metadata-Version: 2.5
2
+ Name: clef-evals
3
+ Version: 0.1.0
4
+ Summary: Calibration-first evaluation toolkit for Cloudflare's Clef decision models. Judge cheap, audit confidence.
5
+ Author: Youssef Ouhaghi Ahmian
6
+ License: Apache-2.0
7
+ Keywords: brier,calibration,clef,cloudflare,ece,evaluation,judge,llm
8
+ Requires-Python: >=3.10
9
+ Requires-Dist: httpx>=0.25
10
+ Requires-Dist: numpy>=1.24
11
+ Provides-Extra: dev
12
+ Requires-Dist: pytest>=7; extra == 'dev'
13
+ Description-Content-Type: text/markdown
14
+
15
+ # clef-evals
16
+
17
+ Calibration-first evaluation toolkit for Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/) decision models.
18
+
19
+ ## What it does
20
+
21
+ - **Judge**: run Clef as an LLM-as-judge over eval sets with typed questions
22
+ - **Calibrate**: compute Expected Calibration Error and Brier score
23
+ - **Gate**: fail CI builds when accuracy or calibration regresses
24
+
25
+ ## Quick start
26
+
27
+ ```bash
28
+ pip install clef-evals
29
+ export CLEF_ACCOUNT_ID=your_id
30
+ export CLEF_API_TOKEN=your_token
31
+ ```
32
+
33
+ ```python
34
+ from clef_evals import ClefJudge, ece, brier_score
35
+
36
+ judge = ClefJudge()
37
+ result = judge.evaluate([
38
+ {"state": "Email: I need a refund", "question": "category",
39
+ "choices": ["billing", "technical", "sales"], "gold": "billing"},
40
+ # ... more items
41
+ ])
42
+ print(f"accuracy: {result.accuracy:.4f}")
43
+ print(f"ece: {result.ece:.4f}")
44
+ print(f"brier: {result.brier:.4f}")
45
+ ```
46
+
47
+ ## License
48
+
49
+ Apache 2.0
@@ -0,0 +1,6 @@
1
+ clef_evals/__init__.py,sha256=ENX20xpEUiPcwzJDsL3z_NnE-5OHOQZbdAISNRtH4jM,3983
2
+ clef_evals/cli.py,sha256=J4-byyJPOfh1HRDNdfQUC30sLdTrEZ0shs3Mb0Aga7w,341
3
+ clef_evals-0.1.0.dist-info/METADATA,sha256=ig2ob-ojUd4TBUb1bPxrLOM4kMOt8YHHxDDD3vdxh6s,1372
4
+ clef_evals-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
5
+ clef_evals-0.1.0.dist-info/entry_points.txt,sha256=5pDsu95VHS4fVd4lqdDQ7TUTsFdfxmebiNcLeawqgD4,50
6
+ clef_evals-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ clef-eval = clef_evals.cli:main