clef-evals 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
publish:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
environment: pypi
|
|
11
|
+
permissions:
|
|
12
|
+
id-token: write
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: astral-sh/setup-uv@v5
|
|
16
|
+
with:
|
|
17
|
+
python-version: "3.12"
|
|
18
|
+
- run: uv build
|
|
19
|
+
- run: uv publish
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: clef-evals
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Calibration-first evaluation toolkit for Cloudflare's Clef decision models. Judge cheap, audit confidence.
|
|
5
|
+
Author: Youssef Ouhaghi Ahmian
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Keywords: brier,calibration,clef,cloudflare,ece,evaluation,judge,llm
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Requires-Dist: httpx>=0.25
|
|
10
|
+
Requires-Dist: numpy>=1.24
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
|
|
15
|
+
# clef-evals
|
|
16
|
+
|
|
17
|
+
Calibration-first evaluation toolkit for Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/) decision models.
|
|
18
|
+
|
|
19
|
+
## What it does
|
|
20
|
+
|
|
21
|
+
- **Judge**: run Clef as an LLM-as-judge over eval sets with typed questions
|
|
22
|
+
- **Calibrate**: compute Expected Calibration Error and Brier score
|
|
23
|
+
- **Gate**: fail CI builds when accuracy or calibration regresses
|
|
24
|
+
|
|
25
|
+
## Quick start
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install clef-evals
|
|
29
|
+
export CLEF_ACCOUNT_ID=your_id
|
|
30
|
+
export CLEF_API_TOKEN=your_token
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from clef_evals import ClefJudge, ece, brier_score
|
|
35
|
+
|
|
36
|
+
judge = ClefJudge()
|
|
37
|
+
result = judge.evaluate([
|
|
38
|
+
{"state": "Email: I need a refund", "question": "category",
|
|
39
|
+
"choices": ["billing", "technical", "sales"], "gold": "billing"},
|
|
40
|
+
# ... more items
|
|
41
|
+
])
|
|
42
|
+
print(f"accuracy: {result.accuracy:.4f}")
|
|
43
|
+
print(f"ece: {result.ece:.4f}")
|
|
44
|
+
print(f"brier: {result.brier:.4f}")
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## License
|
|
48
|
+
|
|
49
|
+
Apache 2.0
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# clef-evals
|
|
2
|
+
|
|
3
|
+
Calibration-first evaluation toolkit for Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/) decision models.
|
|
4
|
+
|
|
5
|
+
## What it does
|
|
6
|
+
|
|
7
|
+
- **Judge**: run Clef as an LLM-as-judge over eval sets with typed questions
|
|
8
|
+
- **Calibrate**: compute Expected Calibration Error and Brier score
|
|
9
|
+
- **Gate**: fail CI builds when accuracy or calibration regresses
|
|
10
|
+
|
|
11
|
+
## Quick start
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
pip install clef-evals
|
|
15
|
+
export CLEF_ACCOUNT_ID=your_id
|
|
16
|
+
export CLEF_API_TOKEN=your_token
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
from clef_evals import ClefJudge, ece, brier_score
|
|
21
|
+
|
|
22
|
+
judge = ClefJudge()
|
|
23
|
+
result = judge.evaluate([
|
|
24
|
+
{"state": "Email: I need a refund", "question": "category",
|
|
25
|
+
"choices": ["billing", "technical", "sales"], "gold": "billing"},
|
|
26
|
+
# ... more items
|
|
27
|
+
])
|
|
28
|
+
print(f"accuracy: {result.accuracy:.4f}")
|
|
29
|
+
print(f"ece: {result.ece:.4f}")
|
|
30
|
+
print(f"brier: {result.brier:.4f}")
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## License
|
|
34
|
+
|
|
35
|
+
Apache 2.0
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "clef-evals"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Calibration-first evaluation toolkit for Cloudflare's Clef decision models. Judge cheap, audit confidence."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = { text = "Apache-2.0" }
|
|
7
|
+
requires-python = ">=3.10"
|
|
8
|
+
authors = [{ name = "Youssef Ouhaghi Ahmian" }]
|
|
9
|
+
keywords = ["clef", "cloudflare", "evaluation", "calibration", "ece", "brier", "llm", "judge"]
|
|
10
|
+
dependencies = ["httpx>=0.25", "numpy>=1.24"]
|
|
11
|
+
|
|
12
|
+
[project.scripts]
|
|
13
|
+
clef-eval = "clef_evals.cli:main"
|
|
14
|
+
|
|
15
|
+
[project.optional-dependencies]
|
|
16
|
+
dev = ["pytest>=7"]
|
|
17
|
+
|
|
18
|
+
[build-system]
|
|
19
|
+
requires = ["hatchling"]
|
|
20
|
+
build-backend = "hatchling.build"
|
|
21
|
+
|
|
22
|
+
[tool.hatch.build.targets.wheel]
|
|
23
|
+
packages = ["src/clef_evals"]
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""clef-evals: Calibration-first evaluation toolkit for Cloudflare's Clef."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def ece(confidences: list[float], correct: list[bool], n_bins: int = 15) -> float:
|
|
12
|
+
"""Expected Calibration Error."""
|
|
13
|
+
confs = np.array(confidences)
|
|
14
|
+
corr = np.array([1.0 if c else 0.0 for c in correct])
|
|
15
|
+
bins = np.linspace(0, 1, n_bins + 1)
|
|
16
|
+
ece_val = 0.0
|
|
17
|
+
for i in range(n_bins):
|
|
18
|
+
mask = (confs > bins[i]) & (confs <= bins[i + 1])
|
|
19
|
+
if mask.sum() == 0:
|
|
20
|
+
continue
|
|
21
|
+
bin_conf = confs[mask].mean()
|
|
22
|
+
bin_acc = corr[mask].mean()
|
|
23
|
+
ece_val += (mask.sum() / len(confs)) * abs(bin_acc - bin_conf)
|
|
24
|
+
return float(ece_val)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def brier_score(confidences: list[float], correct: list[bool]) -> float:
|
|
28
|
+
"""Brier score: mean squared error of probabilistic predictions."""
|
|
29
|
+
confs = np.array(confidences)
|
|
30
|
+
corr = np.array([1.0 if c else 0.0 for c in correct])
|
|
31
|
+
return float(np.mean((confs - corr) ** 2))
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class ClefEvalResult:
|
|
36
|
+
accuracy: float
|
|
37
|
+
ece: float
|
|
38
|
+
brier: float
|
|
39
|
+
n_samples: int
|
|
40
|
+
correct: list[bool]
|
|
41
|
+
confidences: list[float]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass
|
|
45
|
+
class ClefJudge:
|
|
46
|
+
"""Run Clef as an LLM-as-judge over an eval set."""
|
|
47
|
+
|
|
48
|
+
account_id: str = os.environ.get("CLEF_ACCOUNT_ID", "")
|
|
49
|
+
api_token: str = os.environ.get("CLEF_API_TOKEN", "")
|
|
50
|
+
model: str = "@cf/cloudflare/clef-flash"
|
|
51
|
+
base_url: str = "https://api.cloudflare.com/client/v4"
|
|
52
|
+
timeout: float = 60.0
|
|
53
|
+
|
|
54
|
+
def __post_init__(self):
|
|
55
|
+
if not self.account_id:
|
|
56
|
+
raise ValueError("Set CLEF_ACCOUNT_ID")
|
|
57
|
+
|
|
58
|
+
def judge_choice(self, state: str, question: str, choices: list[str]) -> tuple[str, float]:
|
|
59
|
+
"""Ask Clef to choose one option. Returns (choice, confidence)."""
|
|
60
|
+
payload = {
|
|
61
|
+
"state": state,
|
|
62
|
+
"questions": {
|
|
63
|
+
question: {
|
|
64
|
+
"type": "choice",
|
|
65
|
+
"choices": choices,
|
|
66
|
+
},
|
|
67
|
+
"confidence": {
|
|
68
|
+
"type": "score",
|
|
69
|
+
"context": "How confident are you?",
|
|
70
|
+
"levels": ["very_low", "low", "medium", "high", "very_high"],
|
|
71
|
+
},
|
|
72
|
+
},
|
|
73
|
+
}
|
|
74
|
+
resp = httpx.post(
|
|
75
|
+
f"{self.base_url}/accounts/{self.account_id}/ai/run/{self.model}",
|
|
76
|
+
json=payload,
|
|
77
|
+
headers={"Authorization": f"Bearer {self.api_token}"},
|
|
78
|
+
timeout=self.timeout,
|
|
79
|
+
)
|
|
80
|
+
resp.raise_for_status()
|
|
81
|
+
result = resp.json().get("result", {})
|
|
82
|
+
choice = result.get(question, choices[-1])
|
|
83
|
+
conf_map = {"very_low": 0.1, "low": 0.3, "medium": 0.5, "high": 0.7, "very_high": 0.9}
|
|
84
|
+
confidence = conf_map.get(result.get("confidence", "low"), 0.3)
|
|
85
|
+
return choice, confidence
|
|
86
|
+
|
|
87
|
+
def evaluate(
|
|
88
|
+
self,
|
|
89
|
+
eval_set: list[dict],
|
|
90
|
+
state_key: str = "state",
|
|
91
|
+
question_key: str = "question",
|
|
92
|
+
choices_key: str = "choices",
|
|
93
|
+
gold_key: str = "gold",
|
|
94
|
+
) -> ClefEvalResult:
|
|
95
|
+
"""Run judge over an eval set and compute calibration metrics."""
|
|
96
|
+
correct_flags, confidence_list = [], []
|
|
97
|
+
for item in eval_set:
|
|
98
|
+
pred, conf = self.judge_choice(
|
|
99
|
+
item[state_key], item[question_key], item[choices_key]
|
|
100
|
+
)
|
|
101
|
+
is_correct = pred == item[gold_key]
|
|
102
|
+
correct_flags.append(is_correct)
|
|
103
|
+
confidence_list.append(conf)
|
|
104
|
+
|
|
105
|
+
accuracy = sum(correct_flags) / len(correct_flags) if correct_flags else 0.0
|
|
106
|
+
return ClefEvalResult(
|
|
107
|
+
accuracy=accuracy,
|
|
108
|
+
ece=ece(confidence_list, correct_flags),
|
|
109
|
+
brier=brier_score(confidence_list, correct_flags),
|
|
110
|
+
n_samples=len(correct_flags),
|
|
111
|
+
correct=correct_flags,
|
|
112
|
+
confidences=confidence_list,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
__version__ = "0.1.0"
|
|
117
|
+
__all__ = ["ClefJudge", "ClefEvalResult", "ece", "brier_score"]
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""CLI for clef-evals."""
|
|
2
|
+
import argparse
|
|
3
|
+
import json
|
|
4
|
+
|
|
5
|
+
from . import ClefJudge
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def main():
|
|
9
|
+
p = argparse.ArgumentParser(prog="clef-eval", description="Evaluate Clef as judge")
|
|
10
|
+
p.add_argument("--json", action="store_true")
|
|
11
|
+
args = p.parse_args()
|
|
12
|
+
print("See README for usage examples.")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
if __name__ == "__main__":
|
|
16
|
+
main()
|