fastevals 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fastevals/__init__.py +26 -0
- fastevals/cli.py +125 -0
- fastevals/config.py +130 -0
- fastevals/dataset.py +74 -0
- fastevals/evaluators.py +54 -0
- fastevals/exceptions.py +19 -0
- fastevals/mcp_server.py +157 -0
- fastevals/models.py +80 -0
- fastevals/pricing.py +54 -0
- fastevals/providers.py +138 -0
- fastevals/py.typed +0 -0
- fastevals/registry.py +75 -0
- fastevals/report.py +796 -0
- fastevals/runner.py +104 -0
- fastevals/structured.py +115 -0
- fastevals-0.1.0.dist-info/METADATA +244 -0
- fastevals-0.1.0.dist-info/RECORD +21 -0
- fastevals-0.1.0.dist-info/WHEEL +5 -0
- fastevals-0.1.0.dist-info/entry_points.txt +3 -0
- fastevals-0.1.0.dist-info/licenses/LICENSE +21 -0
- fastevals-0.1.0.dist-info/top_level.txt +1 -0
fastevals/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Fast, provider-agnostic LLM evaluation toolkit."""
|
|
2
|
+
|
|
3
|
+
from .config import SUPPORTED_PROVIDERS, ModelSpec, RunConfig
|
|
4
|
+
from .exceptions import ConfigError, FastEvalError, ProviderError, StructuredOutputError
|
|
5
|
+
from .models import ModelResponse, RunResult
|
|
6
|
+
from .registry import load_registry
|
|
7
|
+
from .report import save_report
|
|
8
|
+
from .runner import run
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"SUPPORTED_PROVIDERS",
|
|
14
|
+
"ConfigError",
|
|
15
|
+
"FastEvalError",
|
|
16
|
+
"ModelResponse",
|
|
17
|
+
"ModelSpec",
|
|
18
|
+
"ProviderError",
|
|
19
|
+
"RunConfig",
|
|
20
|
+
"RunResult",
|
|
21
|
+
"StructuredOutputError",
|
|
22
|
+
"__version__",
|
|
23
|
+
"load_registry",
|
|
24
|
+
"run",
|
|
25
|
+
"save_report",
|
|
26
|
+
]
|
fastevals/cli.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""Command-line interface for fastevals."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import asyncio
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from .config import DEFAULT_MAX_CONCURRENCY, SUPPORTED_PROVIDERS, RunConfig
|
|
10
|
+
from .exceptions import FastEvalError
|
|
11
|
+
from .report import save_report
|
|
12
|
+
from .runner import run
|
|
13
|
+
from .structured import shorthand_to_schema
|
|
14
|
+
|
|
15
|
+
ALL_PROVIDERS = "all"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _dotenv_candidates() -> list[Path]:
|
|
19
|
+
return [Path.cwd() / ".env", Path(__file__).resolve().parents[1] / ".env"]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _load_dotenv() -> None:
|
|
23
|
+
"""Load simple KEY=VALUE entries from a project .env if present."""
|
|
24
|
+
for env_path in _dotenv_candidates():
|
|
25
|
+
if not env_path.exists():
|
|
26
|
+
continue
|
|
27
|
+
for raw_line in env_path.read_text().splitlines():
|
|
28
|
+
line = raw_line.strip()
|
|
29
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
30
|
+
continue
|
|
31
|
+
key, value = line.split("=", 1)
|
|
32
|
+
key = key.strip()
|
|
33
|
+
value = value.strip().strip('"').strip("'")
|
|
34
|
+
if key and value and key not in os.environ:
|
|
35
|
+
os.environ[key] = value
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _parse_providers(raw: str) -> frozenset[str]:
|
|
39
|
+
providers = {item.strip().lower() for item in raw.split("|") if item.strip()}
|
|
40
|
+
unknown = sorted(providers - set(SUPPORTED_PROVIDERS) - {ALL_PROVIDERS})
|
|
41
|
+
if unknown:
|
|
42
|
+
supported = ", ".join((*SUPPORTED_PROVIDERS, ALL_PROVIDERS))
|
|
43
|
+
raise argparse.ArgumentTypeError(f"Unknown provider(s): {', '.join(unknown)}. Supported: {supported}")
|
|
44
|
+
return frozenset(providers)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
48
|
+
parser = argparse.ArgumentParser(
|
|
49
|
+
prog="fastevals",
|
|
50
|
+
description="Compare one task results across LLM providers and models.",
|
|
51
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
52
|
+
epilog="""Examples:
|
|
53
|
+
fastevals --prompt \"Summarize this\" --providers \"openai|gemini\" --out runs
|
|
54
|
+
fastevals --image image.png --prompt \"Find widget bboxes\" \\
|
|
55
|
+
--structured-output \"x:int(X coord),y:int(Y coord),width:int(Width),height:int(Height)\" \\
|
|
56
|
+
--providers openai
|
|
57
|
+
fastevals --dataset cases.jsonl --nruns 3 --providers openai --out runs/dataset
|
|
58
|
+
|
|
59
|
+
""",
|
|
60
|
+
)
|
|
61
|
+
parser.add_argument("-p", "--prompt", help="Task prompt (omit when --dataset provides the prompts)")
|
|
62
|
+
parser.add_argument("-s", "--structured-output", help="Structured output compact schema for the response")
|
|
63
|
+
parser.add_argument("-f", "--file", type=Path, help="Input document (sent to the model as an attachment)")
|
|
64
|
+
parser.add_argument("-i", "--image", type=Path, help="Input image")
|
|
65
|
+
parser.add_argument(
|
|
66
|
+
"-pr",
|
|
67
|
+
"--providers",
|
|
68
|
+
type=_parse_providers,
|
|
69
|
+
default=frozenset({ALL_PROVIDERS}),
|
|
70
|
+
help=f"Pipe-separated providers: {'|'.join(SUPPORTED_PROVIDERS)}|all (default: all)",
|
|
71
|
+
)
|
|
72
|
+
parser.add_argument(
|
|
73
|
+
"-r", "--registry", type=Path, help="Path to the model registry TOML (default: config/models.toml)"
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"-d",
|
|
77
|
+
"--dataset",
|
|
78
|
+
type=Path,
|
|
79
|
+
help="JSONL or CSV file with evaluation cases (columns: prompt, expected, evaluator, pattern)",
|
|
80
|
+
)
|
|
81
|
+
parser.add_argument(
|
|
82
|
+
"-n", "--nruns", type=int, default=1, help="Repeat every case this many times for consistency checks"
|
|
83
|
+
)
|
|
84
|
+
parser.add_argument(
|
|
85
|
+
"-c", "--concurrency", type=int, default=DEFAULT_MAX_CONCURRENCY, help="Max parallel model calls"
|
|
86
|
+
)
|
|
87
|
+
parser.add_argument("-o", "--out", type=Path, default=Path("runs"), help="Output directory")
|
|
88
|
+
return parser
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def main(argv: list[str] | None = None) -> int:
|
|
92
|
+
_load_dotenv()
|
|
93
|
+
args = build_parser().parse_args(argv)
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
config = RunConfig(
|
|
97
|
+
prompt=args.prompt or "",
|
|
98
|
+
providers=args.providers,
|
|
99
|
+
file=str(args.file) if args.file else None,
|
|
100
|
+
image=str(args.image) if args.image else None,
|
|
101
|
+
structured_output=shorthand_to_schema(args.structured_output) if args.structured_output else None,
|
|
102
|
+
dataset=str(args.dataset) if args.dataset else None,
|
|
103
|
+
nruns=max(1, args.nruns),
|
|
104
|
+
registry=str(args.registry) if args.registry else None,
|
|
105
|
+
max_concurrency=max(1, args.concurrency),
|
|
106
|
+
out=str(args.out),
|
|
107
|
+
)
|
|
108
|
+
results = asyncio.run(run(config))
|
|
109
|
+
except (FastEvalError, ValueError) as exc:
|
|
110
|
+
print(json.dumps({"ok": False, "error": str(exc), "results": []}, ensure_ascii=False))
|
|
111
|
+
return 1
|
|
112
|
+
|
|
113
|
+
json_path, html_path = save_report(config, results, args.out)
|
|
114
|
+
payload = {
|
|
115
|
+
"ok": all(row.ok for row in results),
|
|
116
|
+
"json_path": str(json_path),
|
|
117
|
+
"html_path": str(html_path) if html_path else None,
|
|
118
|
+
"results": [row.as_dict() for row in results],
|
|
119
|
+
}
|
|
120
|
+
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
121
|
+
return 0 if payload["ok"] else 1
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
if __name__ == "__main__":
|
|
125
|
+
raise SystemExit(main())
|
fastevals/config.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"""Typed configuration objects for runs and model registry entries."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from .exceptions import ConfigError
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"ALL_PROVIDERS",
|
|
11
|
+
"DEFAULT_MAX_CONCURRENCY",
|
|
12
|
+
"DEFAULT_TIMEOUT_S",
|
|
13
|
+
"SUPPORTED_PROVIDERS",
|
|
14
|
+
"ModelSpec",
|
|
15
|
+
"RunConfig",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
ALL_PROVIDERS = "all"
|
|
19
|
+
SUPPORTED_PROVIDERS = ("openai", "gemini", "openrouter")
|
|
20
|
+
DEFAULT_TIMEOUT_S = 120
|
|
21
|
+
DEFAULT_MAX_CONCURRENCY = 4
|
|
22
|
+
DEFAULT_OUT_DIR = "runs"
|
|
23
|
+
MAX_ATTACHMENT_BYTES = 20 * 1024 * 1024
|
|
24
|
+
|
|
25
|
+
_KNOWN_SPEC_KEYS = frozenset(
|
|
26
|
+
{
|
|
27
|
+
"id",
|
|
28
|
+
"provider",
|
|
29
|
+
"model",
|
|
30
|
+
"api_key_env",
|
|
31
|
+
"reasoning_effort",
|
|
32
|
+
"reasoning_efforts",
|
|
33
|
+
"reasoning_parameter",
|
|
34
|
+
"input_cost_usd_per_mtok",
|
|
35
|
+
"cached_input_cost_usd_per_mtok",
|
|
36
|
+
"cached_cost_usd_per_mtok",
|
|
37
|
+
"cached_write_cost_usd_per_mtok",
|
|
38
|
+
"output_cost_usd_per_mtok",
|
|
39
|
+
"reasoning_cost_usd_per_mtok",
|
|
40
|
+
"timeout_s",
|
|
41
|
+
}
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class RunConfig:
|
|
47
|
+
"""Everything needed to execute one evaluation run."""
|
|
48
|
+
|
|
49
|
+
prompt: str
|
|
50
|
+
providers: frozenset[str] = frozenset({ALL_PROVIDERS})
|
|
51
|
+
file: str | None = None
|
|
52
|
+
image: str | None = None
|
|
53
|
+
structured_output: dict[str, Any] | None = None
|
|
54
|
+
dataset: str | None = None
|
|
55
|
+
nruns: int = 1
|
|
56
|
+
registry: str | None = None
|
|
57
|
+
max_concurrency: int = DEFAULT_MAX_CONCURRENCY
|
|
58
|
+
out: str = DEFAULT_OUT_DIR
|
|
59
|
+
|
|
60
|
+
def __post_init__(self) -> None:
|
|
61
|
+
if not self.prompt.strip() and not self.dataset:
|
|
62
|
+
raise ConfigError("Prompt must not be empty when no dataset is given")
|
|
63
|
+
if self.max_concurrency < 1:
|
|
64
|
+
raise ConfigError(f"max_concurrency must be >= 1, got {self.max_concurrency}")
|
|
65
|
+
if self.nruns < 1:
|
|
66
|
+
raise ConfigError(f"nruns must be >= 1, got {self.nruns}")
|
|
67
|
+
for label, path in (("file", self.file), ("image", self.image), ("dataset", self.dataset)):
|
|
68
|
+
if path and not Path(path).exists():
|
|
69
|
+
raise ConfigError(f"{label} not found: {path}")
|
|
70
|
+
unknown = self.requested_providers() - set(SUPPORTED_PROVIDERS) - {ALL_PROVIDERS}
|
|
71
|
+
if unknown:
|
|
72
|
+
supported = ", ".join((*SUPPORTED_PROVIDERS, ALL_PROVIDERS))
|
|
73
|
+
raise ConfigError(f"Unknown provider(s): {', '.join(sorted(unknown))}. Supported: {supported}")
|
|
74
|
+
if self.structured_output is not None:
|
|
75
|
+
schema = self.structured_output
|
|
76
|
+
if not isinstance(schema, dict) or "properties" not in schema:
|
|
77
|
+
raise ConfigError("structured_output must be a JSON Schema object with 'properties'")
|
|
78
|
+
|
|
79
|
+
def requested_providers(self) -> set[str]:
|
|
80
|
+
return {provider.lower() for provider in self.providers}
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@dataclass(frozen=True)
|
|
84
|
+
class ModelSpec:
|
|
85
|
+
"""One concrete provider/model/reasoning combination from the registry."""
|
|
86
|
+
|
|
87
|
+
id: str
|
|
88
|
+
provider: str
|
|
89
|
+
model: str
|
|
90
|
+
api_key_env: str | None = None
|
|
91
|
+
reasoning_effort: str = "off"
|
|
92
|
+
reasoning_parameter: str | None = None
|
|
93
|
+
input_cost_usd_per_mtok: float | None = None
|
|
94
|
+
cached_input_cost_usd_per_mtok: float | None = None
|
|
95
|
+
cached_cost_usd_per_mtok: float | None = None
|
|
96
|
+
cached_write_cost_usd_per_mtok: float | None = None
|
|
97
|
+
output_cost_usd_per_mtok: float | None = None
|
|
98
|
+
reasoning_cost_usd_per_mtok: float | None = None
|
|
99
|
+
timeout_s: int = DEFAULT_TIMEOUT_S
|
|
100
|
+
|
|
101
|
+
@classmethod
|
|
102
|
+
def from_dict(cls, raw: dict[str, Any], spec_id: str) -> "ModelSpec":
|
|
103
|
+
unknown = set(raw) - _KNOWN_SPEC_KEYS - {"type"}
|
|
104
|
+
if unknown:
|
|
105
|
+
raise ConfigError(f"Unknown key(s) in registry entry '{spec_id}': {', '.join(sorted(unknown))}")
|
|
106
|
+
if not isinstance(raw.get("model"), str) or not raw["model"]:
|
|
107
|
+
raise ConfigError(f"Registry entry '{spec_id}' is missing a valid 'model'")
|
|
108
|
+
provider = raw.get("provider", spec_id.split(":", 1)[0])
|
|
109
|
+
if not provider:
|
|
110
|
+
raise ConfigError(f"Registry entry '{spec_id}' is missing a 'provider'")
|
|
111
|
+
fields: dict[str, Any] = {"id": spec_id, "provider": str(provider).lower(), "model": raw["model"]}
|
|
112
|
+
for key in (
|
|
113
|
+
"api_key_env",
|
|
114
|
+
"reasoning_effort",
|
|
115
|
+
"reasoning_parameter",
|
|
116
|
+
"input_cost_usd_per_mtok",
|
|
117
|
+
"cached_input_cost_usd_per_mtok",
|
|
118
|
+
"cached_cost_usd_per_mtok",
|
|
119
|
+
"cached_write_cost_usd_per_mtok",
|
|
120
|
+
"output_cost_usd_per_mtok",
|
|
121
|
+
"reasoning_cost_usd_per_mtok",
|
|
122
|
+
):
|
|
123
|
+
if raw.get(key) is not None:
|
|
124
|
+
fields[key] = raw[key]
|
|
125
|
+
if raw.get("timeout_s") is not None:
|
|
126
|
+
try:
|
|
127
|
+
fields["timeout_s"] = max(1, int(raw["timeout_s"]))
|
|
128
|
+
except (TypeError, ValueError) as exc:
|
|
129
|
+
raise ConfigError(f"Registry entry '{spec_id}' has invalid timeout_s: {raw['timeout_s']!r}") from exc
|
|
130
|
+
return cls(**fields)
|
fastevals/dataset.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Dataset loading: JSONL and CSV evaluation cases."""
|
|
2
|
+
|
|
3
|
+
import csv
|
|
4
|
+
import json
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .exceptions import ConfigError
|
|
9
|
+
|
|
10
|
+
__all__ = ["Case", "load_dataset"]
|
|
11
|
+
|
|
12
|
+
_REQUIRED_FIELDS = frozenset({"prompt"})
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Case:
|
|
17
|
+
"""One evaluation input with optional scoring instructions."""
|
|
18
|
+
|
|
19
|
+
id: str
|
|
20
|
+
prompt: str
|
|
21
|
+
expected: str | None = None
|
|
22
|
+
evaluator: str | None = None
|
|
23
|
+
pattern: str | None = None
|
|
24
|
+
|
|
25
|
+
def as_dict(self) -> dict[str, str | None]:
|
|
26
|
+
return {
|
|
27
|
+
"id": self.id,
|
|
28
|
+
"prompt": self.prompt,
|
|
29
|
+
"expected": self.expected,
|
|
30
|
+
"evaluator": self.evaluator,
|
|
31
|
+
"pattern": self.pattern,
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _make_case(index: int, row: dict[str, str]) -> Case:
|
|
36
|
+
missing = _REQUIRED_FIELDS - {key for key, value in row.items() if value}
|
|
37
|
+
if missing:
|
|
38
|
+
raise ConfigError(f"Dataset row {index + 1} is missing required field(s): {', '.join(sorted(missing))}")
|
|
39
|
+
case_id = str(row.get("id") or f"case-{index + 1:03d}")
|
|
40
|
+
return Case(
|
|
41
|
+
id=case_id,
|
|
42
|
+
prompt=str(row["prompt"]).strip(),
|
|
43
|
+
expected=(str(row["expected"]) if row.get("expected") else None),
|
|
44
|
+
evaluator=(str(row["evaluator"]).strip() or None) if row.get("evaluator") else None,
|
|
45
|
+
pattern=row.get("pattern") or None,
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def load_dataset(path: str | Path) -> list[Case]:
|
|
50
|
+
"""Load cases from a ``.jsonl`` or ``.csv`` file."""
|
|
51
|
+
path = Path(path)
|
|
52
|
+
suffix = path.suffix.lower()
|
|
53
|
+
rows: list[dict[str, str]]
|
|
54
|
+
if suffix == ".jsonl":
|
|
55
|
+
rows = []
|
|
56
|
+
for line_number, line in enumerate(path.read_text().splitlines(), start=1):
|
|
57
|
+
line = line.strip()
|
|
58
|
+
if not line:
|
|
59
|
+
continue
|
|
60
|
+
try:
|
|
61
|
+
item = json.loads(line)
|
|
62
|
+
except json.JSONDecodeError as exc:
|
|
63
|
+
raise ConfigError(f"Dataset line {line_number} is not valid JSON: {exc.msg}") from exc
|
|
64
|
+
if not isinstance(item, dict):
|
|
65
|
+
raise ConfigError(f"Dataset line {line_number} must be a JSON object")
|
|
66
|
+
rows.append({key: "" if value is None else str(value) for key, value in item.items()})
|
|
67
|
+
elif suffix == ".csv":
|
|
68
|
+
with path.open(newline="") as dataset_file:
|
|
69
|
+
rows = [{key: (value or "") for key, value in row.items()} for row in csv.DictReader(dataset_file)]
|
|
70
|
+
else:
|
|
71
|
+
raise ConfigError(f"Unsupported dataset format '{suffix}'. Use .jsonl or .csv")
|
|
72
|
+
if not rows:
|
|
73
|
+
raise ConfigError(f"Dataset is empty: {path}")
|
|
74
|
+
return [_make_case(index, row) for index, row in enumerate(rows)]
|
fastevals/evaluators.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Output evaluators: deterministic scoring for evaluation cases."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from .dataset import Case
|
|
8
|
+
from .exceptions import ConfigError
|
|
9
|
+
|
|
10
|
+
__all__ = ["evaluate_output"]
|
|
11
|
+
|
|
12
|
+
EVALUATORS = ("exact_match", "contains", "json_valid", "regex")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _as_text(output: Any) -> str:
|
|
16
|
+
if isinstance(output, (dict, list)):
|
|
17
|
+
return json.dumps(output, ensure_ascii=False)
|
|
18
|
+
return str(output or "")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def evaluate_output(case: Case, output: Any) -> dict[str, Any]:
|
|
22
|
+
"""Score one output against the case instructions.
|
|
23
|
+
|
|
24
|
+
Returns ``{"evaluator", "passed", "detail"}``; ``passed`` is ``None``
|
|
25
|
+
when the case defines no evaluator.
|
|
26
|
+
"""
|
|
27
|
+
if not case.evaluator:
|
|
28
|
+
return {"evaluator": None, "passed": None, "detail": None}
|
|
29
|
+
|
|
30
|
+
name = case.evaluator.strip().lower()
|
|
31
|
+
text = _as_text(output)
|
|
32
|
+
if name == "exact_match":
|
|
33
|
+
passed = text.strip() == (case.expected or "").strip()
|
|
34
|
+
detail = None if passed else f"expected {case.expected!r}, got {text.strip()[:200]!r}"
|
|
35
|
+
elif name == "contains":
|
|
36
|
+
needle = case.expected or ""
|
|
37
|
+
passed = bool(needle) and needle in text
|
|
38
|
+
detail = None if passed else f"{needle!r} not found in output"
|
|
39
|
+
elif name == "json_valid":
|
|
40
|
+
try:
|
|
41
|
+
json.loads(text)
|
|
42
|
+
passed, detail = True, None
|
|
43
|
+
except json.JSONDecodeError as exc:
|
|
44
|
+
passed, detail = False, f"output is not valid JSON: {exc.msg}"
|
|
45
|
+
elif name == "regex":
|
|
46
|
+
if not case.pattern:
|
|
47
|
+
raise ConfigError(f"Case '{case.id}' uses the regex evaluator but defines no pattern")
|
|
48
|
+
match = re.search(case.pattern, text)
|
|
49
|
+
passed, detail = match is not None, None if match else f"pattern {case.pattern!r} not found"
|
|
50
|
+
else:
|
|
51
|
+
raise ConfigError(
|
|
52
|
+
f"Unknown evaluator '{case.evaluator}' for case '{case.id}'. Supported: {', '.join(EVALUATORS)}"
|
|
53
|
+
)
|
|
54
|
+
return {"evaluator": name, "passed": passed, "detail": detail}
|
fastevals/exceptions.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Exception hierarchy for fastevals."""
|
|
2
|
+
|
|
3
|
+
__all__ = ["ConfigError", "FastEvalError", "ProviderError", "StructuredOutputError"]
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class FastEvalError(Exception):
|
|
7
|
+
"""Base class for all fastevals errors."""
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ConfigError(FastEvalError):
|
|
11
|
+
"""Invalid configuration, registry, or provider selection."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class ProviderError(FastEvalError):
|
|
15
|
+
"""A model call failed: missing credentials, network, or provider error."""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class StructuredOutputError(FastEvalError):
|
|
19
|
+
"""Model output did not satisfy the requested JSON Schema."""
|
fastevals/mcp_server.py
ADDED
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""MCP server exposing fastevals to AI assistants.
|
|
2
|
+
|
|
3
|
+
Run locally with ``fastevals-mcp`` (stdio transport) and register it from any
|
|
4
|
+
MCP client, e.g. Claude Desktop or Claude Code.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import asyncio
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from mcp.server.mcpserver.server import MCPServer
|
|
12
|
+
|
|
13
|
+
from .config import ALL_PROVIDERS, SUPPORTED_PROVIDERS, RunConfig
|
|
14
|
+
from .exceptions import FastEvalError
|
|
15
|
+
from .registry import default_registry_path, load_registry
|
|
16
|
+
from .report import save_report
|
|
17
|
+
from .runner import run
|
|
18
|
+
from .structured import shorthand_to_schema
|
|
19
|
+
|
|
20
|
+
__all__ = ["build_server", "main"]
|
|
21
|
+
|
|
22
|
+
mcp = MCPServer(
|
|
23
|
+
name="fastevals",
|
|
24
|
+
instructions=(
|
|
25
|
+
"fastevals runs one prompt across a matrix of LLM models and providers, "
|
|
26
|
+
"saves every response and returns a comparison summary with cost, "
|
|
27
|
+
"latency and token metrics."
|
|
28
|
+
),
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@mcp.tool()
|
|
33
|
+
async def run_evaluation(
|
|
34
|
+
prompt: str = "",
|
|
35
|
+
providers: str = ALL_PROVIDERS,
|
|
36
|
+
structured_output: str | None = None,
|
|
37
|
+
dataset: str | None = None,
|
|
38
|
+
file: str | None = None,
|
|
39
|
+
image: str | None = None,
|
|
40
|
+
nruns: int = 1,
|
|
41
|
+
out: str = "runs",
|
|
42
|
+
) -> dict[str, Any]:
|
|
43
|
+
"""Run an evaluation matrix and save a JSON + HTML report.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
prompt: The task prompt (omit when ``dataset`` supplies prompts).
|
|
47
|
+
providers: Pipe-separated provider list, e.g. ``openai|openrouter`` or ``all``.
|
|
48
|
+
structured_output: Optional compact schema like ``name:str,age:int``.
|
|
49
|
+
dataset: Optional JSONL/CSV path with cases (prompt, expected, evaluator, pattern).
|
|
50
|
+
file: Optional document attachment (image, PDF or text file).
|
|
51
|
+
image: Optional image attachment.
|
|
52
|
+
nruns: Repeat every case this many times for consistency checks.
|
|
53
|
+
out: Directory where reports are written.
|
|
54
|
+
"""
|
|
55
|
+
try:
|
|
56
|
+
schema = shorthand_to_schema(structured_output) if structured_output else None
|
|
57
|
+
config = RunConfig(
|
|
58
|
+
prompt=prompt,
|
|
59
|
+
providers=frozenset(part.strip().lower() for part in providers.split("|") if part.strip()),
|
|
60
|
+
structured_output=schema,
|
|
61
|
+
dataset=dataset,
|
|
62
|
+
file=file,
|
|
63
|
+
image=image,
|
|
64
|
+
nruns=max(1, nruns),
|
|
65
|
+
out=out,
|
|
66
|
+
)
|
|
67
|
+
results = await run(config)
|
|
68
|
+
json_path, html_path = save_report(config, results, out)
|
|
69
|
+
except FastEvalError as exc:
|
|
70
|
+
return {"ok": False, "error": str(exc)}
|
|
71
|
+
return {
|
|
72
|
+
"ok": all(row.ok for row in results),
|
|
73
|
+
"json_path": str(json_path),
|
|
74
|
+
"html_path": str(html_path),
|
|
75
|
+
"total_cost_usd": sum(row.total_cost_usd or 0 for row in results),
|
|
76
|
+
"results": [
|
|
77
|
+
{
|
|
78
|
+
"case_id": row.case_id,
|
|
79
|
+
"provider": row.provider,
|
|
80
|
+
"model": row.model,
|
|
81
|
+
"reasoning_effort": row.reasoning_effort,
|
|
82
|
+
"latency_ms": row.latency_ms,
|
|
83
|
+
"total_cost_usd": row.total_cost_usd,
|
|
84
|
+
"output": row.output,
|
|
85
|
+
"error": row.error or None,
|
|
86
|
+
"evaluation": row.evaluation,
|
|
87
|
+
}
|
|
88
|
+
for row in results
|
|
89
|
+
],
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@mcp.tool()
|
|
94
|
+
def list_models(registry: str | None = None) -> dict[str, Any]:
|
|
95
|
+
"""List models available in the fastevals registry.
|
|
96
|
+
|
|
97
|
+
Args:
|
|
98
|
+
registry: Optional path to an alternative TOML registry.
|
|
99
|
+
"""
|
|
100
|
+
path = Path(registry) if registry else default_registry_path()
|
|
101
|
+
if not path or not Path(path).exists():
|
|
102
|
+
return {"models": [], "registry": None}
|
|
103
|
+
try:
|
|
104
|
+
entries = load_registry(path)
|
|
105
|
+
except FastEvalError as exc:
|
|
106
|
+
return {"ok": False, "error": str(exc), "models": []}
|
|
107
|
+
return {
|
|
108
|
+
"registry": str(path),
|
|
109
|
+
"supported_providers": list(SUPPORTED_PROVIDERS),
|
|
110
|
+
"models": [
|
|
111
|
+
{
|
|
112
|
+
"id": model_id,
|
|
113
|
+
"provider": entry.get("provider"),
|
|
114
|
+
"model": entry.get("model"),
|
|
115
|
+
"reasoning_efforts": entry.get("reasoning_efforts", entry.get("reasoning_effort", "off")),
|
|
116
|
+
"input_cost_usd_per_mtok": entry.get("input_cost_usd_per_mtok"),
|
|
117
|
+
"output_cost_usd_per_mtok": entry.get("output_cost_usd_per_mtok"),
|
|
118
|
+
}
|
|
119
|
+
for model_id, entry in entries.items()
|
|
120
|
+
],
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@mcp.tool()
|
|
125
|
+
def get_run(json_path: str) -> dict[str, Any]:
|
|
126
|
+
"""Summarize a saved fastevals run from its ``run.json`` file."""
|
|
127
|
+
path = Path(json_path)
|
|
128
|
+
if not path.exists():
|
|
129
|
+
return {"ok": False, "error": f"Run file not found: {json_path}"}
|
|
130
|
+
import json
|
|
131
|
+
|
|
132
|
+
payload = json.loads(path.read_text())
|
|
133
|
+
results: list[dict[str, Any]] = payload.get("results", [])
|
|
134
|
+
scored = [row for row in results if (row.get("evaluation") or {}).get("passed") is not None]
|
|
135
|
+
return {
|
|
136
|
+
"ok": bool(results) and all(not row.get("error") for row in results),
|
|
137
|
+
"created_at": payload.get("created_at"),
|
|
138
|
+
"runs": len(results),
|
|
139
|
+
"errors": sum(1 for row in results if row.get("error")),
|
|
140
|
+
"pass_rate": (sum(1 for row in scored if row["evaluation"]["passed"]) / len(scored) if scored else None),
|
|
141
|
+
"total_cost_usd": sum(row.get("total_cost_usd") or 0 for row in results),
|
|
142
|
+
"html_report": str(path.parent / "report.html"),
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def build_server() -> MCPServer:
|
|
147
|
+
"""Return the configured MCP server instance."""
|
|
148
|
+
return mcp
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def main() -> int:
|
|
152
|
+
asyncio.run(mcp.run_stdio_async())
|
|
153
|
+
return 0
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
if __name__ == "__main__":
|
|
157
|
+
raise SystemExit(main())
|
fastevals/models.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Core result models shared across the package."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import asdict, dataclass, field
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
__all__ = ["ModelResponse", "RunResult"]
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass
|
|
10
|
+
class ModelResponse:
|
|
11
|
+
"""Normalized single-model response returned by provider adapters."""
|
|
12
|
+
|
|
13
|
+
text: str
|
|
14
|
+
input_tokens: int | None = None
|
|
15
|
+
output_tokens: int | None = None
|
|
16
|
+
reasoning_tokens: int | None = None
|
|
17
|
+
cached_tokens: int | None = None
|
|
18
|
+
finish_reason: str | None = None
|
|
19
|
+
response_id: str | None = None
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class RunResult:
|
|
24
|
+
"""Outcome of one cell in the evaluation matrix.
|
|
25
|
+
|
|
26
|
+
Token buckets are disjoint: ``input_tokens`` excludes cached tokens and
|
|
27
|
+
``output_tokens`` excludes reasoning tokens, so each bucket is billed at
|
|
28
|
+
most once.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
provider: str
|
|
32
|
+
model: str
|
|
33
|
+
reasoning_effort: str
|
|
34
|
+
|
|
35
|
+
output: Any
|
|
36
|
+
|
|
37
|
+
case_id: str = "case-001"
|
|
38
|
+
attempt: int = 1
|
|
39
|
+
evaluation: dict[str, Any] | None = field(default=None)
|
|
40
|
+
|
|
41
|
+
time_to_first_token_ms: float | None = None
|
|
42
|
+
latency_ms: float | None = None
|
|
43
|
+
|
|
44
|
+
input_tokens: int | None = None
|
|
45
|
+
output_tokens: int | None = None
|
|
46
|
+
reasoning_tokens: int | None = None
|
|
47
|
+
cached_tokens: int | None = None
|
|
48
|
+
|
|
49
|
+
input_cost_usd: float | None = None
|
|
50
|
+
output_cost_usd: float | None = None
|
|
51
|
+
reasoning_cost_usd: float | None = None
|
|
52
|
+
cached_cost_usd: float | None = None
|
|
53
|
+
|
|
54
|
+
tokens_per_second: float | None = None
|
|
55
|
+
|
|
56
|
+
error: str | None = None
|
|
57
|
+
finish_reason: str | None = None
|
|
58
|
+
response_id: str | None = None
|
|
59
|
+
|
|
60
|
+
def as_dict(self) -> dict[str, Any]:
|
|
61
|
+
"""Serialize including computed convenience fields."""
|
|
62
|
+
data: dict[str, Any] = asdict(self)
|
|
63
|
+
data["ok"] = self.ok
|
|
64
|
+
data["total_cost_usd"] = self.total_cost_usd
|
|
65
|
+
return data
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def ok(self) -> bool:
|
|
69
|
+
return not self.error
|
|
70
|
+
|
|
71
|
+
@property
|
|
72
|
+
def total_cost_usd(self) -> float | None:
|
|
73
|
+
costs = (
|
|
74
|
+
self.input_cost_usd,
|
|
75
|
+
self.output_cost_usd,
|
|
76
|
+
self.reasoning_cost_usd,
|
|
77
|
+
self.cached_cost_usd,
|
|
78
|
+
)
|
|
79
|
+
known_costs = [cost for cost in costs if cost is not None]
|
|
80
|
+
return sum(known_costs) if known_costs else None
|