typesafe-eval 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- typesafe_eval/__init__.py +56 -0
- typesafe_eval/api.py +247 -0
- typesafe_eval/baseline.py +218 -0
- typesafe_eval/cache.py +156 -0
- typesafe_eval/cli.py +880 -0
- typesafe_eval/client.py +1765 -0
- typesafe_eval/exceptions.py +51 -0
- typesafe_eval/hook.py +27 -0
- typesafe_eval/models.py +251 -0
- typesafe_eval/presets/__init__.py +189 -0
- typesafe_eval/presets/design_doc.yaml +68 -0
- typesafe_eval/presets/pr_description.yaml +37 -0
- typesafe_eval/presets/quality.yaml +24 -0
- typesafe_eval/presets/safety.yaml +36 -0
- typesafe_eval/presets/tech_spec.yaml +19 -0
- typesafe_eval/py.typed +1 -0
- typesafe_eval/reporter.py +377 -0
- typesafe_eval/sanitizer.py +956 -0
- typesafe_eval/validator.py +1540 -0
- typesafe_eval-1.0.0.dist-info/METADATA +648 -0
- typesafe_eval-1.0.0.dist-info/RECORD +24 -0
- typesafe_eval-1.0.0.dist-info/WHEEL +4 -0
- typesafe_eval-1.0.0.dist-info/entry_points.txt +4 -0
- typesafe_eval-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""typesafe-eval: Fast, typed multi-dimensional document evaluation CLI using TypeSafe API (Jev)."""
|
|
2
|
+
|
|
3
|
+
from typesafe_eval.api import evaluate, evaluate_document, evaluate_documents
|
|
4
|
+
from typesafe_eval.cache import EvaluationCache
|
|
5
|
+
from typesafe_eval.client import TypeSafeEvaluator
|
|
6
|
+
from typesafe_eval.exceptions import (
|
|
7
|
+
AuthenticationError,
|
|
8
|
+
ConfigurationError,
|
|
9
|
+
ContentViolationError,
|
|
10
|
+
RuntimeEvalError,
|
|
11
|
+
TypeSafeEvalError,
|
|
12
|
+
)
|
|
13
|
+
from typesafe_eval.models import (
|
|
14
|
+
CANDIDATE_DECISION_THRESHOLD,
|
|
15
|
+
NEAR_THRESHOLD_MARGIN,
|
|
16
|
+
ChoiceResult,
|
|
17
|
+
DocumentEvalResult,
|
|
18
|
+
NoulResult,
|
|
19
|
+
PresetConfig,
|
|
20
|
+
QuestionConfig,
|
|
21
|
+
ScoreResult,
|
|
22
|
+
ViolationItem,
|
|
23
|
+
)
|
|
24
|
+
from typesafe_eval.presets import (
|
|
25
|
+
find_project_config,
|
|
26
|
+
load_preset,
|
|
27
|
+
load_project_config,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
__version__ = "1.0.0" # x-release-please-version
|
|
31
|
+
|
|
32
|
+
__all__ = [
|
|
33
|
+
"__version__",
|
|
34
|
+
"evaluate",
|
|
35
|
+
"evaluate_document",
|
|
36
|
+
"evaluate_documents",
|
|
37
|
+
"TypeSafeEvaluator",
|
|
38
|
+
"EvaluationCache",
|
|
39
|
+
"DocumentEvalResult",
|
|
40
|
+
"ViolationItem",
|
|
41
|
+
"ScoreResult",
|
|
42
|
+
"NoulResult",
|
|
43
|
+
"ChoiceResult",
|
|
44
|
+
"PresetConfig",
|
|
45
|
+
"QuestionConfig",
|
|
46
|
+
"load_preset",
|
|
47
|
+
"load_project_config",
|
|
48
|
+
"find_project_config",
|
|
49
|
+
"TypeSafeEvalError",
|
|
50
|
+
"ConfigurationError",
|
|
51
|
+
"AuthenticationError",
|
|
52
|
+
"RuntimeEvalError",
|
|
53
|
+
"ContentViolationError",
|
|
54
|
+
"NEAR_THRESHOLD_MARGIN",
|
|
55
|
+
"CANDIDATE_DECISION_THRESHOLD",
|
|
56
|
+
]
|
typesafe_eval/api.py
ADDED
|
@@ -0,0 +1,247 @@
|
|
|
1
|
+
"""High-level public Python API for typesafe-eval."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from typesafe_eval.client import TypeSafeEvaluator
|
|
11
|
+
from typesafe_eval.exceptions import (
|
|
12
|
+
AuthenticationError,
|
|
13
|
+
ConfigurationError,
|
|
14
|
+
ContentViolationError,
|
|
15
|
+
RuntimeEvalError,
|
|
16
|
+
TypeSafeEvalError,
|
|
17
|
+
)
|
|
18
|
+
from typesafe_eval.models import DocumentEvalResult, PresetConfig
|
|
19
|
+
from typesafe_eval.presets import is_path_excluded, load_preset, load_project_config
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _resolve_preset(
|
|
23
|
+
preset: str | PresetConfig | None = "quality",
|
|
24
|
+
project_root: str | Path | None = None,
|
|
25
|
+
) -> PresetConfig:
|
|
26
|
+
"""Resolves a preset string, PresetConfig, or project configuration."""
|
|
27
|
+
if isinstance(preset, PresetConfig):
|
|
28
|
+
return preset
|
|
29
|
+
|
|
30
|
+
if preset is None or preset == "default":
|
|
31
|
+
start_dir = Path(project_root) if project_root else None
|
|
32
|
+
try:
|
|
33
|
+
discovered_cfg, _ = load_project_config(start_dir=start_dir)
|
|
34
|
+
if discovered_cfg is not None:
|
|
35
|
+
return discovered_cfg
|
|
36
|
+
except Exception as e:
|
|
37
|
+
raise ConfigurationError(f"Error loading discovered project configuration: {e}") from e
|
|
38
|
+
preset = "quality"
|
|
39
|
+
|
|
40
|
+
try:
|
|
41
|
+
return load_preset(preset)
|
|
42
|
+
except Exception as e:
|
|
43
|
+
raise ConfigurationError(f"Failed to load preset '{preset}': {e}") from e
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def evaluate(
|
|
47
|
+
content: str,
|
|
48
|
+
preset: str | PresetConfig = "quality",
|
|
49
|
+
*,
|
|
50
|
+
api_key: str | None = None,
|
|
51
|
+
dry_run: bool = False,
|
|
52
|
+
offline: bool = False,
|
|
53
|
+
cache: bool = True,
|
|
54
|
+
cache_dir: Path | str | None = None,
|
|
55
|
+
mask_secrets: bool = True,
|
|
56
|
+
max_chars: int = 120_000,
|
|
57
|
+
raise_on_violation: bool = False,
|
|
58
|
+
filename: str = "<memory>",
|
|
59
|
+
filepath: str | Path | None = None,
|
|
60
|
+
project_root: str | Path | None = None,
|
|
61
|
+
evaluator: TypeSafeEvaluator | None = None,
|
|
62
|
+
) -> DocumentEvalResult:
|
|
63
|
+
"""Evaluates raw in-memory content against a specified preset."""
|
|
64
|
+
if not isinstance(content, str):
|
|
65
|
+
raise ConfigurationError(f"Expected str content, got {type(content).__name__}")
|
|
66
|
+
|
|
67
|
+
preset_cfg = _resolve_preset(preset, project_root=project_root)
|
|
68
|
+
|
|
69
|
+
resolved_api_key = api_key or os.environ.get("TYPESAFE_API_KEY")
|
|
70
|
+
if not dry_run and not offline and not resolved_api_key:
|
|
71
|
+
raise AuthenticationError(
|
|
72
|
+
"No TypeSafe API key provided. Set the TYPESAFE_API_KEY environment variable "
|
|
73
|
+
"or pass api_key=..."
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
resolved_filepath = str(filepath) if filepath is not None else filename
|
|
77
|
+
active_evaluator = evaluator or TypeSafeEvaluator(
|
|
78
|
+
api_key=resolved_api_key, enable_cache=cache, cache_dir=cache_dir
|
|
79
|
+
)
|
|
80
|
+
try:
|
|
81
|
+
result = active_evaluator.evaluate_content(
|
|
82
|
+
content=content,
|
|
83
|
+
preset=preset_cfg,
|
|
84
|
+
filename=filename,
|
|
85
|
+
filepath=resolved_filepath,
|
|
86
|
+
mask_secrets=mask_secrets,
|
|
87
|
+
max_chars=max_chars,
|
|
88
|
+
dry_run=dry_run,
|
|
89
|
+
offline=offline,
|
|
90
|
+
)
|
|
91
|
+
except TypeSafeEvalError:
|
|
92
|
+
raise
|
|
93
|
+
except Exception as e:
|
|
94
|
+
raise RuntimeEvalError(f"Evaluation failed: {e}") from e
|
|
95
|
+
|
|
96
|
+
if raise_on_violation and not result.passed_thresholds:
|
|
97
|
+
raise ContentViolationError(
|
|
98
|
+
f"Content violation in '{filename}': {'; '.join(result.violations)}",
|
|
99
|
+
violations=result.violations,
|
|
100
|
+
result=result,
|
|
101
|
+
results=[result],
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
return result
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def evaluate_document(
|
|
108
|
+
path: str | Path,
|
|
109
|
+
preset: str | PresetConfig = "quality",
|
|
110
|
+
*,
|
|
111
|
+
api_key: str | None = None,
|
|
112
|
+
dry_run: bool = False,
|
|
113
|
+
offline: bool = False,
|
|
114
|
+
cache: bool = True,
|
|
115
|
+
cache_dir: Path | str | None = None,
|
|
116
|
+
mask_secrets: bool = True,
|
|
117
|
+
max_chars: int = 120_000,
|
|
118
|
+
raise_on_violation: bool = False,
|
|
119
|
+
project_root: str | Path | None = None,
|
|
120
|
+
) -> DocumentEvalResult:
|
|
121
|
+
"""Evaluates a single document file from disk against a specified preset."""
|
|
122
|
+
doc_path = Path(path)
|
|
123
|
+
if not doc_path.is_file():
|
|
124
|
+
raise ConfigurationError(f"Document file not found: {path}")
|
|
125
|
+
|
|
126
|
+
try:
|
|
127
|
+
content = doc_path.read_text(encoding="utf-8")
|
|
128
|
+
except Exception as e:
|
|
129
|
+
raise ConfigurationError(f"Failed to read file '{path}': {e}") from e
|
|
130
|
+
|
|
131
|
+
return evaluate(
|
|
132
|
+
content=content,
|
|
133
|
+
preset=preset,
|
|
134
|
+
api_key=api_key,
|
|
135
|
+
dry_run=dry_run,
|
|
136
|
+
offline=offline,
|
|
137
|
+
cache=cache,
|
|
138
|
+
cache_dir=cache_dir,
|
|
139
|
+
mask_secrets=mask_secrets,
|
|
140
|
+
max_chars=max_chars,
|
|
141
|
+
raise_on_violation=raise_on_violation,
|
|
142
|
+
filename=doc_path.name,
|
|
143
|
+
filepath=str(doc_path),
|
|
144
|
+
project_root=project_root or doc_path.parent,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def evaluate_documents(
|
|
149
|
+
paths: Sequence[str | Path],
|
|
150
|
+
preset: str | PresetConfig = "quality",
|
|
151
|
+
*,
|
|
152
|
+
exclude: Sequence[str] | None = None,
|
|
153
|
+
concurrency: int = 4,
|
|
154
|
+
api_key: str | None = None,
|
|
155
|
+
dry_run: bool = False,
|
|
156
|
+
offline: bool = False,
|
|
157
|
+
cache: bool = True,
|
|
158
|
+
cache_dir: Path | str | None = None,
|
|
159
|
+
mask_secrets: bool = True,
|
|
160
|
+
max_chars: int = 120_000,
|
|
161
|
+
raise_on_violation: bool = False,
|
|
162
|
+
project_root: str | Path | None = None,
|
|
163
|
+
) -> list[DocumentEvalResult]:
|
|
164
|
+
"""Evaluates multiple document files concurrently preserving input order."""
|
|
165
|
+
if not paths:
|
|
166
|
+
return []
|
|
167
|
+
|
|
168
|
+
preset_cfg = _resolve_preset(preset, project_root=project_root)
|
|
169
|
+
|
|
170
|
+
resolved_api_key = api_key or os.environ.get("TYPESAFE_API_KEY")
|
|
171
|
+
if not dry_run and not offline and not resolved_api_key:
|
|
172
|
+
raise AuthenticationError(
|
|
173
|
+
"No TypeSafe API key provided. Set the TYPESAFE_API_KEY environment variable "
|
|
174
|
+
"or pass api_key=..."
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
# Combine exclusion patterns from argument and preset configuration
|
|
178
|
+
combined_excludes = list(exclude or []) + (preset_cfg.exclude or [])
|
|
179
|
+
root = Path(project_root) if project_root else Path.cwd()
|
|
180
|
+
|
|
181
|
+
# Validate all file paths upfront (skipping excluded)
|
|
182
|
+
resolved_paths: list[Path] = []
|
|
183
|
+
for p in paths:
|
|
184
|
+
doc_path = Path(p)
|
|
185
|
+
if combined_excludes and is_path_excluded(doc_path, combined_excludes, root=root):
|
|
186
|
+
continue
|
|
187
|
+
if not doc_path.is_file():
|
|
188
|
+
raise ConfigurationError(f"Document file not found: {p}")
|
|
189
|
+
resolved_paths.append(doc_path)
|
|
190
|
+
|
|
191
|
+
if not resolved_paths:
|
|
192
|
+
return []
|
|
193
|
+
|
|
194
|
+
# Shared evaluator for connection pooling across threads
|
|
195
|
+
shared_evaluator = TypeSafeEvaluator(
|
|
196
|
+
api_key=resolved_api_key, enable_cache=cache, cache_dir=cache_dir
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
def _eval_single(doc_p: Path) -> DocumentEvalResult:
|
|
200
|
+
try:
|
|
201
|
+
content = doc_p.read_text(encoding="utf-8")
|
|
202
|
+
except Exception as e:
|
|
203
|
+
raise ConfigurationError(f"Failed to read file '{doc_p}': {e}") from e
|
|
204
|
+
|
|
205
|
+
return evaluate(
|
|
206
|
+
content=content,
|
|
207
|
+
preset=preset_cfg,
|
|
208
|
+
api_key=resolved_api_key,
|
|
209
|
+
dry_run=dry_run,
|
|
210
|
+
offline=offline,
|
|
211
|
+
cache=cache,
|
|
212
|
+
cache_dir=cache_dir,
|
|
213
|
+
mask_secrets=mask_secrets,
|
|
214
|
+
max_chars=max_chars,
|
|
215
|
+
raise_on_violation=False,
|
|
216
|
+
filename=doc_p.name,
|
|
217
|
+
filepath=str(doc_p),
|
|
218
|
+
project_root=project_root or doc_p.parent,
|
|
219
|
+
evaluator=shared_evaluator,
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
concurrency = max(1, concurrency)
|
|
223
|
+
if concurrency == 1 or len(resolved_paths) == 1:
|
|
224
|
+
results = [_eval_single(p) for p in resolved_paths]
|
|
225
|
+
else:
|
|
226
|
+
with ThreadPoolExecutor(max_workers=min(concurrency, len(resolved_paths))) as executor:
|
|
227
|
+
results = list(executor.map(_eval_single, resolved_paths))
|
|
228
|
+
|
|
229
|
+
if raise_on_violation:
|
|
230
|
+
violated = [r for r in results if not r.passed_thresholds]
|
|
231
|
+
if violated:
|
|
232
|
+
all_violations = [v for r in violated for v in r.violations]
|
|
233
|
+
raise ContentViolationError(
|
|
234
|
+
f"{len(violated)} document(s) failed gates: {'; '.join(all_violations)}",
|
|
235
|
+
violations=all_violations,
|
|
236
|
+
result=violated[0] if len(violated) == 1 else None,
|
|
237
|
+
results=violated,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
return results
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
__all__ = [
|
|
244
|
+
"evaluate",
|
|
245
|
+
"evaluate_document",
|
|
246
|
+
"evaluate_documents",
|
|
247
|
+
]
|
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""Baseline diff mode: loads previous evaluation reports and detects regressions.
|
|
2
|
+
|
|
3
|
+
Implements `--baseline` per Issue #39.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import json
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from typesafe_eval.models import BaselineDiff, DocumentEvalResult, PresetConfig, QuestionDiff
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _norm_preset(name: str | None) -> str:
|
|
13
|
+
return name.lower().replace("-", "_") if name else ""
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _check_comparable(doc: DocumentEvalResult, preset_name: str | None) -> None:
|
|
17
|
+
"""Raises ValueError if a baseline document cannot be compared with a run of preset_name.
|
|
18
|
+
|
|
19
|
+
preset_name=None skips the preset check (the dry-run check always applies).
|
|
20
|
+
"""
|
|
21
|
+
if doc.mock or doc.model == "mock-jev":
|
|
22
|
+
raise ValueError(
|
|
23
|
+
f"Baseline contains dry-run document '{doc.filepath}' (mock=true or model 'mock-jev'); "
|
|
24
|
+
"cannot compare against dry-run baseline."
|
|
25
|
+
)
|
|
26
|
+
if preset_name is not None and _norm_preset(doc.preset_name) != _norm_preset(preset_name):
|
|
27
|
+
raise ValueError(
|
|
28
|
+
f"Baseline document '{doc.filepath}' has preset '{doc.preset_name}', "
|
|
29
|
+
f"differing from current run preset '{preset_name}'."
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def load_baseline(
|
|
34
|
+
baseline_path: str | Path,
|
|
35
|
+
expected_preset: str | None = None,
|
|
36
|
+
) -> dict[str, DocumentEvalResult]:
|
|
37
|
+
"""Loads a previous evaluation report JSON and returns an indexed lookup map."""
|
|
38
|
+
path = Path(baseline_path)
|
|
39
|
+
if not path.is_file():
|
|
40
|
+
raise FileNotFoundError(f"Baseline file not found: {path}")
|
|
41
|
+
|
|
42
|
+
raw_text = path.read_text(encoding="utf-8")
|
|
43
|
+
try:
|
|
44
|
+
data = json.loads(raw_text)
|
|
45
|
+
except Exception as e:
|
|
46
|
+
raise ValueError(f"Failed to parse baseline JSON {path}: {e}") from e
|
|
47
|
+
|
|
48
|
+
if isinstance(data, dict):
|
|
49
|
+
items = [data]
|
|
50
|
+
elif isinstance(data, list):
|
|
51
|
+
items = data
|
|
52
|
+
else:
|
|
53
|
+
raise ValueError(
|
|
54
|
+
f"Invalid baseline report structure in {path}: expected JSON list or object"
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
lookup: dict[str, DocumentEvalResult] = {}
|
|
58
|
+
for item in items:
|
|
59
|
+
try:
|
|
60
|
+
doc_res = DocumentEvalResult(**item)
|
|
61
|
+
except Exception as e:
|
|
62
|
+
raise ValueError(f"Invalid document result in baseline {path}: {e}") from e
|
|
63
|
+
|
|
64
|
+
_check_comparable(doc_res, expected_preset)
|
|
65
|
+
|
|
66
|
+
# Index by filepath and resolved path only (do not index by filename to prevent collision)
|
|
67
|
+
lookup[doc_res.filepath] = doc_res
|
|
68
|
+
try:
|
|
69
|
+
resolved_key = str(Path(doc_res.filepath).resolve())
|
|
70
|
+
lookup[resolved_key] = doc_res
|
|
71
|
+
except Exception:
|
|
72
|
+
pass
|
|
73
|
+
|
|
74
|
+
return lookup
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def compare_document_with_baseline(
|
|
78
|
+
result: DocumentEvalResult,
|
|
79
|
+
baseline_lookup: dict[str, DocumentEvalResult],
|
|
80
|
+
preset: PresetConfig,
|
|
81
|
+
default_max_drop: float = 0.10,
|
|
82
|
+
) -> tuple[DocumentEvalResult, bool, str | None]:
|
|
83
|
+
"""Compares a current document result with its baseline counterpart.
|
|
84
|
+
|
|
85
|
+
Returns (updated_result, has_regression, optional_truncation_warning).
|
|
86
|
+
"""
|
|
87
|
+
# 1. Direct match by exact filepath or resolved absolute path
|
|
88
|
+
prev = None
|
|
89
|
+
for key in (result.filepath, str(Path(result.filepath).resolve())):
|
|
90
|
+
if key in baseline_lookup:
|
|
91
|
+
prev = baseline_lookup[key]
|
|
92
|
+
break
|
|
93
|
+
|
|
94
|
+
# 2. Portable cross-environment relative / suffix matching (e.g. CI runner absolute path vs local run)
|
|
95
|
+
if prev is None:
|
|
96
|
+
norm_target = Path(result.filepath).as_posix().lstrip("./")
|
|
97
|
+
seen_ids = set()
|
|
98
|
+
norm_docs = []
|
|
99
|
+
for doc in baseline_lookup.values():
|
|
100
|
+
if id(doc) not in seen_ids:
|
|
101
|
+
seen_ids.add(id(doc))
|
|
102
|
+
norm_docs.append((doc, Path(doc.filepath).as_posix().lstrip("./")))
|
|
103
|
+
|
|
104
|
+
# 2a. Prioritize exact normalized match
|
|
105
|
+
exact_matches = [doc for doc, doc_norm in norm_docs if doc_norm == norm_target]
|
|
106
|
+
if len(exact_matches) == 1:
|
|
107
|
+
prev = exact_matches[0]
|
|
108
|
+
elif not exact_matches:
|
|
109
|
+
# 2b. Fallback to suffix matching
|
|
110
|
+
suffix_matches = [
|
|
111
|
+
doc
|
|
112
|
+
for doc, doc_norm in norm_docs
|
|
113
|
+
if doc_norm.endswith("/" + norm_target) or norm_target.endswith("/" + doc_norm)
|
|
114
|
+
]
|
|
115
|
+
if len(suffix_matches) == 1:
|
|
116
|
+
prev = suffix_matches[0]
|
|
117
|
+
|
|
118
|
+
if prev is None:
|
|
119
|
+
result.baseline_diff = BaselineDiff(status="new")
|
|
120
|
+
return result, False, None
|
|
121
|
+
|
|
122
|
+
_check_comparable(prev, preset.name)
|
|
123
|
+
|
|
124
|
+
# Check truncation mismatch
|
|
125
|
+
trunc_mismatch = prev.was_truncated != result.was_truncated
|
|
126
|
+
warning_msg = None
|
|
127
|
+
if trunc_mismatch:
|
|
128
|
+
warning_msg = (
|
|
129
|
+
f"Truncation status differs for {result.filename} "
|
|
130
|
+
f"(baseline was_truncated={prev.was_truncated}, current was_truncated={result.was_truncated}). "
|
|
131
|
+
f"Scores may be shifted by truncation."
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
has_regression = False
|
|
135
|
+
diff_questions: dict[str, QuestionDiff] = {}
|
|
136
|
+
|
|
137
|
+
for q_id, q_cfg in preset.questions.items():
|
|
138
|
+
threshold = q_cfg.max_drop if (q_cfg and q_cfg.max_drop is not None) else default_max_drop
|
|
139
|
+
|
|
140
|
+
curr_val: float | None = None
|
|
141
|
+
prev_val: float | None = None
|
|
142
|
+
|
|
143
|
+
if q_id in result.scores and q_id in prev.scores:
|
|
144
|
+
curr_val = result.scores[q_id].normalized_score
|
|
145
|
+
prev_val = prev.scores[q_id].normalized_score
|
|
146
|
+
elif q_id in result.nouls and q_id in prev.nouls:
|
|
147
|
+
if (
|
|
148
|
+
result.nouls[q_id].probability is not None
|
|
149
|
+
and prev.nouls[q_id].probability is not None
|
|
150
|
+
):
|
|
151
|
+
curr_val = result.nouls[q_id].probability
|
|
152
|
+
prev_val = prev.nouls[q_id].probability
|
|
153
|
+
|
|
154
|
+
if curr_val is not None and prev_val is not None:
|
|
155
|
+
delta = curr_val - prev_val
|
|
156
|
+
is_risk_question = bool(q_cfg and q_cfg.max_threshold is not None)
|
|
157
|
+
|
|
158
|
+
if is_risk_question:
|
|
159
|
+
# For risk questions with max_threshold (e.g. has_secrets, has_pii, confidentiality_risk):
|
|
160
|
+
# A rise in probability/risk is a regression
|
|
161
|
+
rise = curr_val - prev_val
|
|
162
|
+
regressed = rise > threshold
|
|
163
|
+
|
|
164
|
+
diff_questions[q_id] = QuestionDiff(
|
|
165
|
+
previous=prev_val,
|
|
166
|
+
current=curr_val,
|
|
167
|
+
delta=delta,
|
|
168
|
+
max_drop=threshold,
|
|
169
|
+
regressed=regressed,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
is_advisory = bool(q_cfg and q_cfg.advisory)
|
|
173
|
+
if regressed:
|
|
174
|
+
msg = (
|
|
175
|
+
f"Baseline risk rise: '{q_id}' rose by {rise:.2f} "
|
|
176
|
+
f"(prev: {prev_val:.2f}, curr: {curr_val:.2f}, max allowed rise: {threshold:.2f})"
|
|
177
|
+
)
|
|
178
|
+
if is_advisory:
|
|
179
|
+
result.warnings.append(f"{msg} (warning)")
|
|
180
|
+
else:
|
|
181
|
+
has_regression = True
|
|
182
|
+
result.passed_thresholds = False
|
|
183
|
+
result.violations.append(msg)
|
|
184
|
+
else:
|
|
185
|
+
# For standard/min_threshold questions (e.g. clarity, completeness):
|
|
186
|
+
# A drop in score is a regression
|
|
187
|
+
drop = prev_val - curr_val
|
|
188
|
+
regressed = drop > threshold
|
|
189
|
+
|
|
190
|
+
diff_questions[q_id] = QuestionDiff(
|
|
191
|
+
previous=prev_val,
|
|
192
|
+
current=curr_val,
|
|
193
|
+
delta=delta,
|
|
194
|
+
max_drop=threshold,
|
|
195
|
+
regressed=regressed,
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
is_advisory = bool(q_cfg and q_cfg.advisory)
|
|
199
|
+
if regressed:
|
|
200
|
+
msg = (
|
|
201
|
+
f"Baseline drop: '{q_id}' dropped by {drop:.2f} "
|
|
202
|
+
f"(prev: {prev_val:.2f}, curr: {curr_val:.2f}, max allowed drop: {threshold:.2f})"
|
|
203
|
+
)
|
|
204
|
+
if is_advisory:
|
|
205
|
+
result.warnings.append(f"{msg} (warning)")
|
|
206
|
+
else:
|
|
207
|
+
has_regression = True
|
|
208
|
+
result.passed_thresholds = False
|
|
209
|
+
result.violations.append(msg)
|
|
210
|
+
|
|
211
|
+
result.baseline_diff = BaselineDiff(
|
|
212
|
+
status="compared",
|
|
213
|
+
baseline_filepath=prev.filepath,
|
|
214
|
+
truncation_mismatch=trunc_mismatch,
|
|
215
|
+
questions=diff_questions,
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
return result, has_regression, warning_msg
|
typesafe_eval/cache.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Content-addressable evaluation result cache."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from typesafe_eval.models import DocumentEvalResult, PresetConfig
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _get_version() -> str:
|
|
14
|
+
try:
|
|
15
|
+
import typesafe_eval
|
|
16
|
+
|
|
17
|
+
return getattr(typesafe_eval, "__version__", "1.0.0")
|
|
18
|
+
except Exception:
|
|
19
|
+
return "1.0.0"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def get_default_cache_dir() -> Path:
|
|
23
|
+
"""Returns the default directory for caching evaluation results."""
|
|
24
|
+
env_dir = os.environ.get("TYPESAFE_CACHE_DIR")
|
|
25
|
+
if env_dir:
|
|
26
|
+
return Path(env_dir)
|
|
27
|
+
|
|
28
|
+
if os.environ.get("GITHUB_ACTIONS") == "true":
|
|
29
|
+
return Path(".typesafe-eval-cache")
|
|
30
|
+
|
|
31
|
+
# Standard XDG cache directory
|
|
32
|
+
xdg_cache = os.environ.get("XDG_CACHE_HOME")
|
|
33
|
+
if xdg_cache:
|
|
34
|
+
target = Path(xdg_cache) / "typesafe-eval"
|
|
35
|
+
else:
|
|
36
|
+
target = Path.home() / ".cache" / "typesafe-eval"
|
|
37
|
+
|
|
38
|
+
try:
|
|
39
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
40
|
+
return target
|
|
41
|
+
except OSError:
|
|
42
|
+
# Fallback to local workspace cache directory if home/xdg cache is not writable
|
|
43
|
+
return Path(".typesafe-eval-cache")
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class EvaluationCache:
|
|
47
|
+
"""Local file-based content-addressable cache for document evaluations."""
|
|
48
|
+
|
|
49
|
+
def __init__(
|
|
50
|
+
self,
|
|
51
|
+
cache_dir: Path | str | None = None,
|
|
52
|
+
enabled: bool = True,
|
|
53
|
+
) -> None:
|
|
54
|
+
self.cache_dir = Path(cache_dir) if cache_dir else get_default_cache_dir()
|
|
55
|
+
self.enabled = enabled
|
|
56
|
+
|
|
57
|
+
def compute_key(
|
|
58
|
+
self,
|
|
59
|
+
content: str,
|
|
60
|
+
preset: PresetConfig,
|
|
61
|
+
mask_secrets: bool = True,
|
|
62
|
+
max_chars: int = 25000,
|
|
63
|
+
) -> str:
|
|
64
|
+
"""Computes a deterministic SHA-256 hash from document content, preset rules, and evaluation options."""
|
|
65
|
+
preset_repr = preset.model_dump_json()
|
|
66
|
+
hasher = hashlib.sha256()
|
|
67
|
+
hasher.update(content.encode())
|
|
68
|
+
hasher.update(b"::")
|
|
69
|
+
hasher.update(preset_repr.encode())
|
|
70
|
+
hasher.update(b"::")
|
|
71
|
+
hasher.update(f"mask={mask_secrets}:max={max_chars}".encode())
|
|
72
|
+
hasher.update(b"::")
|
|
73
|
+
hasher.update(_get_version().encode())
|
|
74
|
+
return hasher.hexdigest()
|
|
75
|
+
|
|
76
|
+
def get(
|
|
77
|
+
self,
|
|
78
|
+
content: str,
|
|
79
|
+
preset: PresetConfig,
|
|
80
|
+
mask_secrets: bool = True,
|
|
81
|
+
max_chars: int = 25000,
|
|
82
|
+
) -> DocumentEvalResult | None:
|
|
83
|
+
"""Retrieves a cached evaluation result if present and valid."""
|
|
84
|
+
if not self.enabled:
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
key = self.compute_key(content, preset, mask_secrets=mask_secrets, max_chars=max_chars)
|
|
88
|
+
cache_file = self.cache_dir / f"{key}.json"
|
|
89
|
+
|
|
90
|
+
if not cache_file.is_file():
|
|
91
|
+
return None
|
|
92
|
+
|
|
93
|
+
try:
|
|
94
|
+
raw_text = cache_file.read_text(encoding="utf-8")
|
|
95
|
+
result = DocumentEvalResult.model_validate_json(raw_text)
|
|
96
|
+
result.cached = True
|
|
97
|
+
result.api_calls = 0
|
|
98
|
+
return result
|
|
99
|
+
except Exception:
|
|
100
|
+
# Corrupted cache file; safely remove and treat as miss
|
|
101
|
+
try:
|
|
102
|
+
cache_file.unlink(missing_ok=True)
|
|
103
|
+
except OSError:
|
|
104
|
+
pass
|
|
105
|
+
return None
|
|
106
|
+
|
|
107
|
+
def set(
|
|
108
|
+
self,
|
|
109
|
+
content: str,
|
|
110
|
+
preset: PresetConfig,
|
|
111
|
+
result: DocumentEvalResult,
|
|
112
|
+
mask_secrets: bool = True,
|
|
113
|
+
max_chars: int = 25000,
|
|
114
|
+
) -> None:
|
|
115
|
+
"""Stores an evaluation result in the cache atomically."""
|
|
116
|
+
if not self.enabled:
|
|
117
|
+
return
|
|
118
|
+
|
|
119
|
+
try:
|
|
120
|
+
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
|
121
|
+
key = self.compute_key(content, preset, mask_secrets=mask_secrets, max_chars=max_chars)
|
|
122
|
+
target_path = self.cache_dir / f"{key}.json"
|
|
123
|
+
temp_path = self.cache_dir / f"{key}.tmp.{os.getpid()}"
|
|
124
|
+
|
|
125
|
+
# Clone result to avoid mutating caller object
|
|
126
|
+
res_dict = result.model_dump()
|
|
127
|
+
res_dict["cached"] = False # Stored template is pristine
|
|
128
|
+
json_str = json.dumps(res_dict, ensure_ascii=False)
|
|
129
|
+
|
|
130
|
+
temp_path.write_text(json_str, encoding="utf-8")
|
|
131
|
+
temp_path.replace(target_path)
|
|
132
|
+
except Exception:
|
|
133
|
+
# Cache failure must never block application execution
|
|
134
|
+
pass
|
|
135
|
+
|
|
136
|
+
def clear(self) -> int:
|
|
137
|
+
"""Clears all cached evaluations and returns the count of removed files."""
|
|
138
|
+
if not self.cache_dir.is_dir():
|
|
139
|
+
return 0
|
|
140
|
+
|
|
141
|
+
count = 0
|
|
142
|
+
try:
|
|
143
|
+
for item in self.cache_dir.glob("*.json"):
|
|
144
|
+
try:
|
|
145
|
+
item.unlink()
|
|
146
|
+
count += 1
|
|
147
|
+
except OSError:
|
|
148
|
+
pass
|
|
149
|
+
for item in self.cache_dir.glob("*.tmp.*"):
|
|
150
|
+
try:
|
|
151
|
+
item.unlink()
|
|
152
|
+
except OSError:
|
|
153
|
+
pass
|
|
154
|
+
except OSError:
|
|
155
|
+
pass
|
|
156
|
+
return count
|