fastevals 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
fastevals/__init__.py ADDED
@@ -0,0 +1,26 @@
1
+ """Fast, provider-agnostic LLM evaluation toolkit."""
2
+
3
+ from .config import SUPPORTED_PROVIDERS, ModelSpec, RunConfig
4
+ from .exceptions import ConfigError, FastEvalError, ProviderError, StructuredOutputError
5
+ from .models import ModelResponse, RunResult
6
+ from .registry import load_registry
7
+ from .report import save_report
8
+ from .runner import run
9
+
10
+ __version__ = "0.1.0"
11
+
12
+ __all__ = [
13
+ "SUPPORTED_PROVIDERS",
14
+ "ConfigError",
15
+ "FastEvalError",
16
+ "ModelResponse",
17
+ "ModelSpec",
18
+ "ProviderError",
19
+ "RunConfig",
20
+ "RunResult",
21
+ "StructuredOutputError",
22
+ "__version__",
23
+ "load_registry",
24
+ "run",
25
+ "save_report",
26
+ ]
fastevals/cli.py ADDED
@@ -0,0 +1,125 @@
1
+ """Command-line interface for fastevals."""
2
+
3
+ import argparse
4
+ import asyncio
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+
9
+ from .config import DEFAULT_MAX_CONCURRENCY, SUPPORTED_PROVIDERS, RunConfig
10
+ from .exceptions import FastEvalError
11
+ from .report import save_report
12
+ from .runner import run
13
+ from .structured import shorthand_to_schema
14
+
15
+ ALL_PROVIDERS = "all"
16
+
17
+
18
+ def _dotenv_candidates() -> list[Path]:
19
+ return [Path.cwd() / ".env", Path(__file__).resolve().parents[1] / ".env"]
20
+
21
+
22
+ def _load_dotenv() -> None:
23
+ """Load simple KEY=VALUE entries from a project .env if present."""
24
+ for env_path in _dotenv_candidates():
25
+ if not env_path.exists():
26
+ continue
27
+ for raw_line in env_path.read_text().splitlines():
28
+ line = raw_line.strip()
29
+ if not line or line.startswith("#") or "=" not in line:
30
+ continue
31
+ key, value = line.split("=", 1)
32
+ key = key.strip()
33
+ value = value.strip().strip('"').strip("'")
34
+ if key and value and key not in os.environ:
35
+ os.environ[key] = value
36
+
37
+
38
+ def _parse_providers(raw: str) -> frozenset[str]:
39
+ providers = {item.strip().lower() for item in raw.split("|") if item.strip()}
40
+ unknown = sorted(providers - set(SUPPORTED_PROVIDERS) - {ALL_PROVIDERS})
41
+ if unknown:
42
+ supported = ", ".join((*SUPPORTED_PROVIDERS, ALL_PROVIDERS))
43
+ raise argparse.ArgumentTypeError(f"Unknown provider(s): {', '.join(unknown)}. Supported: {supported}")
44
+ return frozenset(providers)
45
+
46
+
47
+ def build_parser() -> argparse.ArgumentParser:
48
+ parser = argparse.ArgumentParser(
49
+ prog="fastevals",
50
+ description="Compare one task results across LLM providers and models.",
51
+ formatter_class=argparse.RawDescriptionHelpFormatter,
52
+ epilog="""Examples:
53
+ fastevals --prompt \"Summarize this\" --providers \"openai|gemini\" --out runs
54
+ fastevals --image image.png --prompt \"Find widget bboxes\" \\
55
+ --structured-output \"x:int(X coord),y:int(Y coord),width:int(Width),height:int(Height)\" \\
56
+ --providers openai
57
+ fastevals --dataset cases.jsonl --nruns 3 --providers openai --out runs/dataset
58
+
59
+ """,
60
+ )
61
+ parser.add_argument("-p", "--prompt", help="Task prompt (omit when --dataset provides the prompts)")
62
+ parser.add_argument("-s", "--structured-output", help="Structured output compact schema for the response")
63
+ parser.add_argument("-f", "--file", type=Path, help="Input document (sent to the model as an attachment)")
64
+ parser.add_argument("-i", "--image", type=Path, help="Input image")
65
+ parser.add_argument(
66
+ "-pr",
67
+ "--providers",
68
+ type=_parse_providers,
69
+ default=frozenset({ALL_PROVIDERS}),
70
+ help=f"Pipe-separated providers: {'|'.join(SUPPORTED_PROVIDERS)}|all (default: all)",
71
+ )
72
+ parser.add_argument(
73
+ "-r", "--registry", type=Path, help="Path to the model registry TOML (default: config/models.toml)"
74
+ )
75
+ parser.add_argument(
76
+ "-d",
77
+ "--dataset",
78
+ type=Path,
79
+ help="JSONL or CSV file with evaluation cases (columns: prompt, expected, evaluator, pattern)",
80
+ )
81
+ parser.add_argument(
82
+ "-n", "--nruns", type=int, default=1, help="Repeat every case this many times for consistency checks"
83
+ )
84
+ parser.add_argument(
85
+ "-c", "--concurrency", type=int, default=DEFAULT_MAX_CONCURRENCY, help="Max parallel model calls"
86
+ )
87
+ parser.add_argument("-o", "--out", type=Path, default=Path("runs"), help="Output directory")
88
+ return parser
89
+
90
+
91
+ def main(argv: list[str] | None = None) -> int:
92
+ _load_dotenv()
93
+ args = build_parser().parse_args(argv)
94
+
95
+ try:
96
+ config = RunConfig(
97
+ prompt=args.prompt or "",
98
+ providers=args.providers,
99
+ file=str(args.file) if args.file else None,
100
+ image=str(args.image) if args.image else None,
101
+ structured_output=shorthand_to_schema(args.structured_output) if args.structured_output else None,
102
+ dataset=str(args.dataset) if args.dataset else None,
103
+ nruns=max(1, args.nruns),
104
+ registry=str(args.registry) if args.registry else None,
105
+ max_concurrency=max(1, args.concurrency),
106
+ out=str(args.out),
107
+ )
108
+ results = asyncio.run(run(config))
109
+ except (FastEvalError, ValueError) as exc:
110
+ print(json.dumps({"ok": False, "error": str(exc), "results": []}, ensure_ascii=False))
111
+ return 1
112
+
113
+ json_path, html_path = save_report(config, results, args.out)
114
+ payload = {
115
+ "ok": all(row.ok for row in results),
116
+ "json_path": str(json_path),
117
+ "html_path": str(html_path) if html_path else None,
118
+ "results": [row.as_dict() for row in results],
119
+ }
120
+ print(json.dumps(payload, ensure_ascii=False, indent=2))
121
+ return 0 if payload["ok"] else 1
122
+
123
+
124
+ if __name__ == "__main__":
125
+ raise SystemExit(main())
fastevals/config.py ADDED
@@ -0,0 +1,130 @@
1
+ """Typed configuration objects for runs and model registry entries."""
2
+
3
+ from dataclasses import dataclass
4
+ from pathlib import Path
5
+ from typing import Any
6
+
7
+ from .exceptions import ConfigError
8
+
9
+ __all__ = [
10
+ "ALL_PROVIDERS",
11
+ "DEFAULT_MAX_CONCURRENCY",
12
+ "DEFAULT_TIMEOUT_S",
13
+ "SUPPORTED_PROVIDERS",
14
+ "ModelSpec",
15
+ "RunConfig",
16
+ ]
17
+
18
+ ALL_PROVIDERS = "all"
19
+ SUPPORTED_PROVIDERS = ("openai", "gemini", "openrouter")
20
+ DEFAULT_TIMEOUT_S = 120
21
+ DEFAULT_MAX_CONCURRENCY = 4
22
+ DEFAULT_OUT_DIR = "runs"
23
+ MAX_ATTACHMENT_BYTES = 20 * 1024 * 1024
24
+
25
+ _KNOWN_SPEC_KEYS = frozenset(
26
+ {
27
+ "id",
28
+ "provider",
29
+ "model",
30
+ "api_key_env",
31
+ "reasoning_effort",
32
+ "reasoning_efforts",
33
+ "reasoning_parameter",
34
+ "input_cost_usd_per_mtok",
35
+ "cached_input_cost_usd_per_mtok",
36
+ "cached_cost_usd_per_mtok",
37
+ "cached_write_cost_usd_per_mtok",
38
+ "output_cost_usd_per_mtok",
39
+ "reasoning_cost_usd_per_mtok",
40
+ "timeout_s",
41
+ }
42
+ )
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class RunConfig:
47
+ """Everything needed to execute one evaluation run."""
48
+
49
+ prompt: str
50
+ providers: frozenset[str] = frozenset({ALL_PROVIDERS})
51
+ file: str | None = None
52
+ image: str | None = None
53
+ structured_output: dict[str, Any] | None = None
54
+ dataset: str | None = None
55
+ nruns: int = 1
56
+ registry: str | None = None
57
+ max_concurrency: int = DEFAULT_MAX_CONCURRENCY
58
+ out: str = DEFAULT_OUT_DIR
59
+
60
+ def __post_init__(self) -> None:
61
+ if not self.prompt.strip() and not self.dataset:
62
+ raise ConfigError("Prompt must not be empty when no dataset is given")
63
+ if self.max_concurrency < 1:
64
+ raise ConfigError(f"max_concurrency must be >= 1, got {self.max_concurrency}")
65
+ if self.nruns < 1:
66
+ raise ConfigError(f"nruns must be >= 1, got {self.nruns}")
67
+ for label, path in (("file", self.file), ("image", self.image), ("dataset", self.dataset)):
68
+ if path and not Path(path).exists():
69
+ raise ConfigError(f"{label} not found: {path}")
70
+ unknown = self.requested_providers() - set(SUPPORTED_PROVIDERS) - {ALL_PROVIDERS}
71
+ if unknown:
72
+ supported = ", ".join((*SUPPORTED_PROVIDERS, ALL_PROVIDERS))
73
+ raise ConfigError(f"Unknown provider(s): {', '.join(sorted(unknown))}. Supported: {supported}")
74
+ if self.structured_output is not None:
75
+ schema = self.structured_output
76
+ if not isinstance(schema, dict) or "properties" not in schema:
77
+ raise ConfigError("structured_output must be a JSON Schema object with 'properties'")
78
+
79
+ def requested_providers(self) -> set[str]:
80
+ return {provider.lower() for provider in self.providers}
81
+
82
+
83
+ @dataclass(frozen=True)
84
+ class ModelSpec:
85
+ """One concrete provider/model/reasoning combination from the registry."""
86
+
87
+ id: str
88
+ provider: str
89
+ model: str
90
+ api_key_env: str | None = None
91
+ reasoning_effort: str = "off"
92
+ reasoning_parameter: str | None = None
93
+ input_cost_usd_per_mtok: float | None = None
94
+ cached_input_cost_usd_per_mtok: float | None = None
95
+ cached_cost_usd_per_mtok: float | None = None
96
+ cached_write_cost_usd_per_mtok: float | None = None
97
+ output_cost_usd_per_mtok: float | None = None
98
+ reasoning_cost_usd_per_mtok: float | None = None
99
+ timeout_s: int = DEFAULT_TIMEOUT_S
100
+
101
+ @classmethod
102
+ def from_dict(cls, raw: dict[str, Any], spec_id: str) -> "ModelSpec":
103
+ unknown = set(raw) - _KNOWN_SPEC_KEYS - {"type"}
104
+ if unknown:
105
+ raise ConfigError(f"Unknown key(s) in registry entry '{spec_id}': {', '.join(sorted(unknown))}")
106
+ if not isinstance(raw.get("model"), str) or not raw["model"]:
107
+ raise ConfigError(f"Registry entry '{spec_id}' is missing a valid 'model'")
108
+ provider = raw.get("provider", spec_id.split(":", 1)[0])
109
+ if not provider:
110
+ raise ConfigError(f"Registry entry '{spec_id}' is missing a 'provider'")
111
+ fields: dict[str, Any] = {"id": spec_id, "provider": str(provider).lower(), "model": raw["model"]}
112
+ for key in (
113
+ "api_key_env",
114
+ "reasoning_effort",
115
+ "reasoning_parameter",
116
+ "input_cost_usd_per_mtok",
117
+ "cached_input_cost_usd_per_mtok",
118
+ "cached_cost_usd_per_mtok",
119
+ "cached_write_cost_usd_per_mtok",
120
+ "output_cost_usd_per_mtok",
121
+ "reasoning_cost_usd_per_mtok",
122
+ ):
123
+ if raw.get(key) is not None:
124
+ fields[key] = raw[key]
125
+ if raw.get("timeout_s") is not None:
126
+ try:
127
+ fields["timeout_s"] = max(1, int(raw["timeout_s"]))
128
+ except (TypeError, ValueError) as exc:
129
+ raise ConfigError(f"Registry entry '{spec_id}' has invalid timeout_s: {raw['timeout_s']!r}") from exc
130
+ return cls(**fields)
fastevals/dataset.py ADDED
@@ -0,0 +1,74 @@
1
+ """Dataset loading: JSONL and CSV evaluation cases."""
2
+
3
+ import csv
4
+ import json
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+
8
+ from .exceptions import ConfigError
9
+
10
+ __all__ = ["Case", "load_dataset"]
11
+
12
+ _REQUIRED_FIELDS = frozenset({"prompt"})
13
+
14
+
15
+ @dataclass(frozen=True)
16
+ class Case:
17
+ """One evaluation input with optional scoring instructions."""
18
+
19
+ id: str
20
+ prompt: str
21
+ expected: str | None = None
22
+ evaluator: str | None = None
23
+ pattern: str | None = None
24
+
25
+ def as_dict(self) -> dict[str, str | None]:
26
+ return {
27
+ "id": self.id,
28
+ "prompt": self.prompt,
29
+ "expected": self.expected,
30
+ "evaluator": self.evaluator,
31
+ "pattern": self.pattern,
32
+ }
33
+
34
+
35
+ def _make_case(index: int, row: dict[str, str]) -> Case:
36
+ missing = _REQUIRED_FIELDS - {key for key, value in row.items() if value}
37
+ if missing:
38
+ raise ConfigError(f"Dataset row {index + 1} is missing required field(s): {', '.join(sorted(missing))}")
39
+ case_id = str(row.get("id") or f"case-{index + 1:03d}")
40
+ return Case(
41
+ id=case_id,
42
+ prompt=str(row["prompt"]).strip(),
43
+ expected=(str(row["expected"]) if row.get("expected") else None),
44
+ evaluator=(str(row["evaluator"]).strip() or None) if row.get("evaluator") else None,
45
+ pattern=row.get("pattern") or None,
46
+ )
47
+
48
+
49
+ def load_dataset(path: str | Path) -> list[Case]:
50
+ """Load cases from a ``.jsonl`` or ``.csv`` file."""
51
+ path = Path(path)
52
+ suffix = path.suffix.lower()
53
+ rows: list[dict[str, str]]
54
+ if suffix == ".jsonl":
55
+ rows = []
56
+ for line_number, line in enumerate(path.read_text().splitlines(), start=1):
57
+ line = line.strip()
58
+ if not line:
59
+ continue
60
+ try:
61
+ item = json.loads(line)
62
+ except json.JSONDecodeError as exc:
63
+ raise ConfigError(f"Dataset line {line_number} is not valid JSON: {exc.msg}") from exc
64
+ if not isinstance(item, dict):
65
+ raise ConfigError(f"Dataset line {line_number} must be a JSON object")
66
+ rows.append({key: "" if value is None else str(value) for key, value in item.items()})
67
+ elif suffix == ".csv":
68
+ with path.open(newline="") as dataset_file:
69
+ rows = [{key: (value or "") for key, value in row.items()} for row in csv.DictReader(dataset_file)]
70
+ else:
71
+ raise ConfigError(f"Unsupported dataset format '{suffix}'. Use .jsonl or .csv")
72
+ if not rows:
73
+ raise ConfigError(f"Dataset is empty: {path}")
74
+ return [_make_case(index, row) for index, row in enumerate(rows)]
@@ -0,0 +1,54 @@
1
+ """Output evaluators: deterministic scoring for evaluation cases."""
2
+
3
+ import json
4
+ import re
5
+ from typing import Any
6
+
7
+ from .dataset import Case
8
+ from .exceptions import ConfigError
9
+
10
+ __all__ = ["evaluate_output"]
11
+
12
+ EVALUATORS = ("exact_match", "contains", "json_valid", "regex")
13
+
14
+
15
+ def _as_text(output: Any) -> str:
16
+ if isinstance(output, (dict, list)):
17
+ return json.dumps(output, ensure_ascii=False)
18
+ return str(output or "")
19
+
20
+
21
+ def evaluate_output(case: Case, output: Any) -> dict[str, Any]:
22
+ """Score one output against the case instructions.
23
+
24
+ Returns ``{"evaluator", "passed", "detail"}``; ``passed`` is ``None``
25
+ when the case defines no evaluator.
26
+ """
27
+ if not case.evaluator:
28
+ return {"evaluator": None, "passed": None, "detail": None}
29
+
30
+ name = case.evaluator.strip().lower()
31
+ text = _as_text(output)
32
+ if name == "exact_match":
33
+ passed = text.strip() == (case.expected or "").strip()
34
+ detail = None if passed else f"expected {case.expected!r}, got {text.strip()[:200]!r}"
35
+ elif name == "contains":
36
+ needle = case.expected or ""
37
+ passed = bool(needle) and needle in text
38
+ detail = None if passed else f"{needle!r} not found in output"
39
+ elif name == "json_valid":
40
+ try:
41
+ json.loads(text)
42
+ passed, detail = True, None
43
+ except json.JSONDecodeError as exc:
44
+ passed, detail = False, f"output is not valid JSON: {exc.msg}"
45
+ elif name == "regex":
46
+ if not case.pattern:
47
+ raise ConfigError(f"Case '{case.id}' uses the regex evaluator but defines no pattern")
48
+ match = re.search(case.pattern, text)
49
+ passed, detail = match is not None, None if match else f"pattern {case.pattern!r} not found"
50
+ else:
51
+ raise ConfigError(
52
+ f"Unknown evaluator '{case.evaluator}' for case '{case.id}'. Supported: {', '.join(EVALUATORS)}"
53
+ )
54
+ return {"evaluator": name, "passed": passed, "detail": detail}
@@ -0,0 +1,19 @@
1
+ """Exception hierarchy for fastevals."""
2
+
3
+ __all__ = ["ConfigError", "FastEvalError", "ProviderError", "StructuredOutputError"]
4
+
5
+
6
+ class FastEvalError(Exception):
7
+ """Base class for all fastevals errors."""
8
+
9
+
10
+ class ConfigError(FastEvalError):
11
+ """Invalid configuration, registry, or provider selection."""
12
+
13
+
14
+ class ProviderError(FastEvalError):
15
+ """A model call failed: missing credentials, network, or provider error."""
16
+
17
+
18
+ class StructuredOutputError(FastEvalError):
19
+ """Model output did not satisfy the requested JSON Schema."""
@@ -0,0 +1,157 @@
1
+ """MCP server exposing fastevals to AI assistants.
2
+
3
+ Run locally with ``fastevals-mcp`` (stdio transport) and register it from any
4
+ MCP client, e.g. Claude Desktop or Claude Code.
5
+ """
6
+
7
+ import asyncio
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ from mcp.server.mcpserver.server import MCPServer
12
+
13
+ from .config import ALL_PROVIDERS, SUPPORTED_PROVIDERS, RunConfig
14
+ from .exceptions import FastEvalError
15
+ from .registry import default_registry_path, load_registry
16
+ from .report import save_report
17
+ from .runner import run
18
+ from .structured import shorthand_to_schema
19
+
20
+ __all__ = ["build_server", "main"]
21
+
22
+ mcp = MCPServer(
23
+ name="fastevals",
24
+ instructions=(
25
+ "fastevals runs one prompt across a matrix of LLM models and providers, "
26
+ "saves every response and returns a comparison summary with cost, "
27
+ "latency and token metrics."
28
+ ),
29
+ )
30
+
31
+
32
+ @mcp.tool()
33
+ async def run_evaluation(
34
+ prompt: str = "",
35
+ providers: str = ALL_PROVIDERS,
36
+ structured_output: str | None = None,
37
+ dataset: str | None = None,
38
+ file: str | None = None,
39
+ image: str | None = None,
40
+ nruns: int = 1,
41
+ out: str = "runs",
42
+ ) -> dict[str, Any]:
43
+ """Run an evaluation matrix and save a JSON + HTML report.
44
+
45
+ Args:
46
+ prompt: The task prompt (omit when ``dataset`` supplies prompts).
47
+ providers: Pipe-separated provider list, e.g. ``openai|openrouter`` or ``all``.
48
+ structured_output: Optional compact schema like ``name:str,age:int``.
49
+ dataset: Optional JSONL/CSV path with cases (prompt, expected, evaluator, pattern).
50
+ file: Optional document attachment (image, PDF or text file).
51
+ image: Optional image attachment.
52
+ nruns: Repeat every case this many times for consistency checks.
53
+ out: Directory where reports are written.
54
+ """
55
+ try:
56
+ schema = shorthand_to_schema(structured_output) if structured_output else None
57
+ config = RunConfig(
58
+ prompt=prompt,
59
+ providers=frozenset(part.strip().lower() for part in providers.split("|") if part.strip()),
60
+ structured_output=schema,
61
+ dataset=dataset,
62
+ file=file,
63
+ image=image,
64
+ nruns=max(1, nruns),
65
+ out=out,
66
+ )
67
+ results = await run(config)
68
+ json_path, html_path = save_report(config, results, out)
69
+ except FastEvalError as exc:
70
+ return {"ok": False, "error": str(exc)}
71
+ return {
72
+ "ok": all(row.ok for row in results),
73
+ "json_path": str(json_path),
74
+ "html_path": str(html_path),
75
+ "total_cost_usd": sum(row.total_cost_usd or 0 for row in results),
76
+ "results": [
77
+ {
78
+ "case_id": row.case_id,
79
+ "provider": row.provider,
80
+ "model": row.model,
81
+ "reasoning_effort": row.reasoning_effort,
82
+ "latency_ms": row.latency_ms,
83
+ "total_cost_usd": row.total_cost_usd,
84
+ "output": row.output,
85
+ "error": row.error or None,
86
+ "evaluation": row.evaluation,
87
+ }
88
+ for row in results
89
+ ],
90
+ }
91
+
92
+
93
+ @mcp.tool()
94
+ def list_models(registry: str | None = None) -> dict[str, Any]:
95
+ """List models available in the fastevals registry.
96
+
97
+ Args:
98
+ registry: Optional path to an alternative TOML registry.
99
+ """
100
+ path = Path(registry) if registry else default_registry_path()
101
+ if not path or not Path(path).exists():
102
+ return {"models": [], "registry": None}
103
+ try:
104
+ entries = load_registry(path)
105
+ except FastEvalError as exc:
106
+ return {"ok": False, "error": str(exc), "models": []}
107
+ return {
108
+ "registry": str(path),
109
+ "supported_providers": list(SUPPORTED_PROVIDERS),
110
+ "models": [
111
+ {
112
+ "id": model_id,
113
+ "provider": entry.get("provider"),
114
+ "model": entry.get("model"),
115
+ "reasoning_efforts": entry.get("reasoning_efforts", entry.get("reasoning_effort", "off")),
116
+ "input_cost_usd_per_mtok": entry.get("input_cost_usd_per_mtok"),
117
+ "output_cost_usd_per_mtok": entry.get("output_cost_usd_per_mtok"),
118
+ }
119
+ for model_id, entry in entries.items()
120
+ ],
121
+ }
122
+
123
+
124
+ @mcp.tool()
125
+ def get_run(json_path: str) -> dict[str, Any]:
126
+ """Summarize a saved fastevals run from its ``run.json`` file."""
127
+ path = Path(json_path)
128
+ if not path.exists():
129
+ return {"ok": False, "error": f"Run file not found: {json_path}"}
130
+ import json
131
+
132
+ payload = json.loads(path.read_text())
133
+ results: list[dict[str, Any]] = payload.get("results", [])
134
+ scored = [row for row in results if (row.get("evaluation") or {}).get("passed") is not None]
135
+ return {
136
+ "ok": bool(results) and all(not row.get("error") for row in results),
137
+ "created_at": payload.get("created_at"),
138
+ "runs": len(results),
139
+ "errors": sum(1 for row in results if row.get("error")),
140
+ "pass_rate": (sum(1 for row in scored if row["evaluation"]["passed"]) / len(scored) if scored else None),
141
+ "total_cost_usd": sum(row.get("total_cost_usd") or 0 for row in results),
142
+ "html_report": str(path.parent / "report.html"),
143
+ }
144
+
145
+
146
+ def build_server() -> MCPServer:
147
+ """Return the configured MCP server instance."""
148
+ return mcp
149
+
150
+
151
+ def main() -> int:
152
+ asyncio.run(mcp.run_stdio_async())
153
+ return 0
154
+
155
+
156
+ if __name__ == "__main__":
157
+ raise SystemExit(main())
fastevals/models.py ADDED
@@ -0,0 +1,80 @@
1
+ """Core result models shared across the package."""
2
+
3
+ from dataclasses import asdict, dataclass, field
4
+ from typing import Any
5
+
6
+ __all__ = ["ModelResponse", "RunResult"]
7
+
8
+
9
+ @dataclass
10
+ class ModelResponse:
11
+ """Normalized single-model response returned by provider adapters."""
12
+
13
+ text: str
14
+ input_tokens: int | None = None
15
+ output_tokens: int | None = None
16
+ reasoning_tokens: int | None = None
17
+ cached_tokens: int | None = None
18
+ finish_reason: str | None = None
19
+ response_id: str | None = None
20
+
21
+
22
+ @dataclass
23
+ class RunResult:
24
+ """Outcome of one cell in the evaluation matrix.
25
+
26
+ Token buckets are disjoint: ``input_tokens`` excludes cached tokens and
27
+ ``output_tokens`` excludes reasoning tokens, so each bucket is billed at
28
+ most once.
29
+ """
30
+
31
+ provider: str
32
+ model: str
33
+ reasoning_effort: str
34
+
35
+ output: Any
36
+
37
+ case_id: str = "case-001"
38
+ attempt: int = 1
39
+ evaluation: dict[str, Any] | None = field(default=None)
40
+
41
+ time_to_first_token_ms: float | None = None
42
+ latency_ms: float | None = None
43
+
44
+ input_tokens: int | None = None
45
+ output_tokens: int | None = None
46
+ reasoning_tokens: int | None = None
47
+ cached_tokens: int | None = None
48
+
49
+ input_cost_usd: float | None = None
50
+ output_cost_usd: float | None = None
51
+ reasoning_cost_usd: float | None = None
52
+ cached_cost_usd: float | None = None
53
+
54
+ tokens_per_second: float | None = None
55
+
56
+ error: str | None = None
57
+ finish_reason: str | None = None
58
+ response_id: str | None = None
59
+
60
+ def as_dict(self) -> dict[str, Any]:
61
+ """Serialize including computed convenience fields."""
62
+ data: dict[str, Any] = asdict(self)
63
+ data["ok"] = self.ok
64
+ data["total_cost_usd"] = self.total_cost_usd
65
+ return data
66
+
67
+ @property
68
+ def ok(self) -> bool:
69
+ return not self.error
70
+
71
+ @property
72
+ def total_cost_usd(self) -> float | None:
73
+ costs = (
74
+ self.input_cost_usd,
75
+ self.output_cost_usd,
76
+ self.reasoning_cost_usd,
77
+ self.cached_cost_usd,
78
+ )
79
+ known_costs = [cost for cost in costs if cost is not None]
80
+ return sum(known_costs) if known_costs else None