typesafe-eval 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- typesafe_eval/__init__.py +56 -0
- typesafe_eval/api.py +247 -0
- typesafe_eval/baseline.py +218 -0
- typesafe_eval/cache.py +156 -0
- typesafe_eval/cli.py +880 -0
- typesafe_eval/client.py +1765 -0
- typesafe_eval/exceptions.py +51 -0
- typesafe_eval/hook.py +27 -0
- typesafe_eval/models.py +251 -0
- typesafe_eval/presets/__init__.py +189 -0
- typesafe_eval/presets/design_doc.yaml +68 -0
- typesafe_eval/presets/pr_description.yaml +37 -0
- typesafe_eval/presets/quality.yaml +24 -0
- typesafe_eval/presets/safety.yaml +36 -0
- typesafe_eval/presets/tech_spec.yaml +19 -0
- typesafe_eval/py.typed +1 -0
- typesafe_eval/reporter.py +377 -0
- typesafe_eval/sanitizer.py +956 -0
- typesafe_eval/validator.py +1540 -0
- typesafe_eval-1.0.0.dist-info/METADATA +648 -0
- typesafe_eval-1.0.0.dist-info/RECORD +24 -0
- typesafe_eval-1.0.0.dist-info/WHEEL +4 -0
- typesafe_eval-1.0.0.dist-info/entry_points.txt +4 -0
- typesafe_eval-1.0.0.dist-info/licenses/LICENSE +21 -0
typesafe_eval/cli.py
ADDED
|
@@ -0,0 +1,880 @@
|
|
|
1
|
+
"""Command line interface for TypeSafe document evaluation and validation."""
|
|
2
|
+
|
|
3
|
+
import glob
|
|
4
|
+
import os
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
import click
|
|
11
|
+
from rich.console import Console
|
|
12
|
+
from rich.markup import escape
|
|
13
|
+
|
|
14
|
+
from typesafe_eval import __version__
|
|
15
|
+
from typesafe_eval.baseline import compare_document_with_baseline, load_baseline
|
|
16
|
+
from typesafe_eval.client import TypeSafeEvaluator
|
|
17
|
+
from typesafe_eval.models import DocumentEvalResult, PresetConfig
|
|
18
|
+
from typesafe_eval.presets import (
|
|
19
|
+
is_default_ignored,
|
|
20
|
+
is_path_excluded,
|
|
21
|
+
list_builtin_presets,
|
|
22
|
+
load_preset,
|
|
23
|
+
load_project_config,
|
|
24
|
+
)
|
|
25
|
+
from typesafe_eval.reporter import (
|
|
26
|
+
render_github_annotations,
|
|
27
|
+
render_json,
|
|
28
|
+
render_markdown,
|
|
29
|
+
render_table,
|
|
30
|
+
)
|
|
31
|
+
from typesafe_eval.validator import (
|
|
32
|
+
generate_ablation_variants,
|
|
33
|
+
load_labels_file,
|
|
34
|
+
render_validation_json,
|
|
35
|
+
render_validation_markdown,
|
|
36
|
+
render_validation_table,
|
|
37
|
+
run_validation,
|
|
38
|
+
validate_labels_preset,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
err_console = Console(stderr=True)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def get_git_changed_files(staged: bool = False, since: str | None = None) -> list[str]:
|
|
45
|
+
"""Retrieves list of modified/added files from git."""
|
|
46
|
+
git_env = dict(os.environ)
|
|
47
|
+
if "GIT_CONFIG_GLOBAL" not in git_env:
|
|
48
|
+
git_env["GIT_CONFIG_GLOBAL"] = "/dev/null"
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
root_proc = subprocess.run(
|
|
52
|
+
["git", "rev-parse", "--show-toplevel"],
|
|
53
|
+
capture_output=True,
|
|
54
|
+
text=True,
|
|
55
|
+
check=True,
|
|
56
|
+
env=git_env,
|
|
57
|
+
)
|
|
58
|
+
repo_root = Path(root_proc.stdout.strip())
|
|
59
|
+
except subprocess.CalledProcessError as e:
|
|
60
|
+
err_msg = e.stderr.strip() if e.stderr else str(e)
|
|
61
|
+
raise RuntimeError(f"Not a git repository: {err_msg}") from e
|
|
62
|
+
except FileNotFoundError as e:
|
|
63
|
+
raise RuntimeError("git executable not found in PATH") from e
|
|
64
|
+
|
|
65
|
+
cmd = ["git", "diff", "--name-only", "--diff-filter=ACMR"]
|
|
66
|
+
if staged:
|
|
67
|
+
cmd.append("--cached")
|
|
68
|
+
if since:
|
|
69
|
+
cmd.append(since)
|
|
70
|
+
|
|
71
|
+
try:
|
|
72
|
+
proc = subprocess.run(cmd, capture_output=True, text=True, check=True, env=git_env)
|
|
73
|
+
files = [line.strip() for line in proc.stdout.splitlines() if line.strip()]
|
|
74
|
+
result = []
|
|
75
|
+
for f in files:
|
|
76
|
+
full_path = (repo_root / f).resolve()
|
|
77
|
+
if full_path.is_file():
|
|
78
|
+
result.append(str(full_path))
|
|
79
|
+
return result
|
|
80
|
+
except subprocess.CalledProcessError as e:
|
|
81
|
+
err_msg = e.stderr.strip() if e.stderr else str(e)
|
|
82
|
+
raise RuntimeError(f"Git diff extraction failed: {err_msg}") from e
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
DEFAULT_EVAL_EXTENSIONS = {
|
|
86
|
+
".md",
|
|
87
|
+
".markdown",
|
|
88
|
+
".mdown",
|
|
89
|
+
".txt",
|
|
90
|
+
".text",
|
|
91
|
+
".rst",
|
|
92
|
+
".adoc",
|
|
93
|
+
".asciidoc",
|
|
94
|
+
".json",
|
|
95
|
+
".yaml",
|
|
96
|
+
".yml",
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _emit_empty_diff_result(
|
|
101
|
+
output_format: str,
|
|
102
|
+
out: Path | None,
|
|
103
|
+
msg: str = "No modified or staged files matched evaluation criteria.",
|
|
104
|
+
) -> None:
|
|
105
|
+
"""Emits clean result when no modified or staged files match criteria."""
|
|
106
|
+
if output_format == "json":
|
|
107
|
+
out_content = render_json([])
|
|
108
|
+
if out:
|
|
109
|
+
out.write_text(out_content, encoding="utf-8")
|
|
110
|
+
else:
|
|
111
|
+
click.echo(out_content)
|
|
112
|
+
else:
|
|
113
|
+
if out:
|
|
114
|
+
out.write_text(msg + "\n", encoding="utf-8")
|
|
115
|
+
else:
|
|
116
|
+
click.echo(msg)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class DefaultGroup(click.Group):
|
|
120
|
+
"""Click Group that defaults to a specified command if no subcommand matches."""
|
|
121
|
+
|
|
122
|
+
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
|
123
|
+
self.default_cmd_name: str | None = kwargs.pop("default_if_no_match", None)
|
|
124
|
+
super().__init__(*args, **kwargs)
|
|
125
|
+
|
|
126
|
+
def parse_args(self, ctx: click.Context, args: list[str]) -> list[str]:
|
|
127
|
+
if not args:
|
|
128
|
+
if self.default_cmd_name:
|
|
129
|
+
args = [self.default_cmd_name]
|
|
130
|
+
return super().parse_args(ctx, args)
|
|
131
|
+
cmd_name = args[0]
|
|
132
|
+
if cmd_name not in self.commands:
|
|
133
|
+
if self.default_cmd_name and cmd_name not in ("--help", "-h", "--version"):
|
|
134
|
+
args.insert(0, self.default_cmd_name)
|
|
135
|
+
return super().parse_args(ctx, args)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@click.group(
|
|
139
|
+
name="typesafe-eval",
|
|
140
|
+
cls=DefaultGroup,
|
|
141
|
+
default_if_no_match="eval",
|
|
142
|
+
context_settings={"help_option_names": ["-h", "--help"]},
|
|
143
|
+
)
|
|
144
|
+
@click.version_option(version=__version__, prog_name="typesafe-eval")
|
|
145
|
+
def main() -> None:
|
|
146
|
+
"""Fast, typed multi-dimensional document evaluation CLI using TypeSafe API (Jev)."""
|
|
147
|
+
pass
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@main.command(name="eval", context_settings={"help_option_names": ["-h", "--help"]})
|
|
151
|
+
@click.argument("files", nargs=-1, type=str)
|
|
152
|
+
@click.option(
|
|
153
|
+
"-p",
|
|
154
|
+
"--preset",
|
|
155
|
+
default=None,
|
|
156
|
+
help=f"Built-in preset to use ({', '.join(list_builtin_presets())}). Default: project config or quality.",
|
|
157
|
+
)
|
|
158
|
+
@click.option(
|
|
159
|
+
"-c",
|
|
160
|
+
"--config",
|
|
161
|
+
type=click.Path(exists=True, dir_okay=False, path_type=Path),
|
|
162
|
+
help="Custom YAML configuration file defining evaluation dimensions.",
|
|
163
|
+
)
|
|
164
|
+
@click.option(
|
|
165
|
+
"-f",
|
|
166
|
+
"--format",
|
|
167
|
+
"output_format",
|
|
168
|
+
type=click.Choice(["table", "json", "markdown", "github"], case_sensitive=False),
|
|
169
|
+
default="table",
|
|
170
|
+
help="Output presentation format. Default: table.",
|
|
171
|
+
)
|
|
172
|
+
@click.option(
|
|
173
|
+
"-o",
|
|
174
|
+
"--out",
|
|
175
|
+
type=click.Path(dir_okay=False, path_type=Path),
|
|
176
|
+
help="Save report output to specified file path.",
|
|
177
|
+
)
|
|
178
|
+
@click.option(
|
|
179
|
+
"--mask-secrets/--no-mask-secrets",
|
|
180
|
+
"--mask/--no-mask",
|
|
181
|
+
default=True,
|
|
182
|
+
help="Automatically redact detected API keys, credentials, and PII before API call. Default: enabled.",
|
|
183
|
+
)
|
|
184
|
+
@click.option(
|
|
185
|
+
"--max-chars",
|
|
186
|
+
type=click.IntRange(min=1),
|
|
187
|
+
default=25000,
|
|
188
|
+
help="Maximum character threshold before safe head/tail truncation. Default: 25000.",
|
|
189
|
+
)
|
|
190
|
+
@click.option(
|
|
191
|
+
"--dry-run",
|
|
192
|
+
is_flag=True,
|
|
193
|
+
help="Validate files and inputs using mock results without sending requests to TypeSafe API.",
|
|
194
|
+
)
|
|
195
|
+
@click.option(
|
|
196
|
+
"--offline",
|
|
197
|
+
"--rules-only",
|
|
198
|
+
is_flag=True,
|
|
199
|
+
default=False,
|
|
200
|
+
help="Run static regex rule checks offline without TypeSafe API (no API key required).",
|
|
201
|
+
)
|
|
202
|
+
@click.option(
|
|
203
|
+
"--api-key",
|
|
204
|
+
envvar="TYPESAFE_API_KEY",
|
|
205
|
+
help="TypeSafe API key (falls back to TYPESAFE_API_KEY environment variable).",
|
|
206
|
+
)
|
|
207
|
+
@click.option(
|
|
208
|
+
"--fail-on-threshold/--no-fail-on-threshold",
|
|
209
|
+
default=True,
|
|
210
|
+
help="Exit with non-zero status code (exit 1) if any threshold violation occurs. Default: enabled.",
|
|
211
|
+
)
|
|
212
|
+
@click.option(
|
|
213
|
+
"--baseline",
|
|
214
|
+
type=click.Path(dir_okay=False, path_type=Path),
|
|
215
|
+
help="Previous JSON report to compare against and detect score regressions.",
|
|
216
|
+
)
|
|
217
|
+
@click.option(
|
|
218
|
+
"-j",
|
|
219
|
+
"--concurrency",
|
|
220
|
+
type=click.IntRange(min=1),
|
|
221
|
+
default=4,
|
|
222
|
+
help="Number of concurrent worker threads for parallel document evaluation. Default: 4.",
|
|
223
|
+
)
|
|
224
|
+
@click.option(
|
|
225
|
+
"--staged",
|
|
226
|
+
is_flag=True,
|
|
227
|
+
help="Evaluate only files staged for git commit.",
|
|
228
|
+
)
|
|
229
|
+
@click.option(
|
|
230
|
+
"--since",
|
|
231
|
+
"--changed-since",
|
|
232
|
+
"changed_since",
|
|
233
|
+
type=str,
|
|
234
|
+
default=None,
|
|
235
|
+
help="Evaluate files modified or added in git since the specified commit or branch reference.",
|
|
236
|
+
)
|
|
237
|
+
@click.option(
|
|
238
|
+
"--list-presets",
|
|
239
|
+
is_flag=True,
|
|
240
|
+
help="List all available built-in evaluation presets and exit.",
|
|
241
|
+
)
|
|
242
|
+
@click.option(
|
|
243
|
+
"--cache/--no-cache",
|
|
244
|
+
default=True,
|
|
245
|
+
help="Enable or disable evaluation result cache. Default: enabled.",
|
|
246
|
+
)
|
|
247
|
+
@click.option(
|
|
248
|
+
"--cache-dir",
|
|
249
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
250
|
+
default=None,
|
|
251
|
+
help="Custom directory for caching evaluation results.",
|
|
252
|
+
)
|
|
253
|
+
@click.option(
|
|
254
|
+
"-e",
|
|
255
|
+
"--exclude",
|
|
256
|
+
"exclude_patterns",
|
|
257
|
+
multiple=True,
|
|
258
|
+
type=str,
|
|
259
|
+
help="Glob pattern(s) to exclude from evaluation (can be specified multiple times).",
|
|
260
|
+
)
|
|
261
|
+
def eval_command(
|
|
262
|
+
files: list[str],
|
|
263
|
+
preset: str | None,
|
|
264
|
+
config: Path | None,
|
|
265
|
+
output_format: str,
|
|
266
|
+
out: Path | None,
|
|
267
|
+
mask_secrets: bool,
|
|
268
|
+
max_chars: int,
|
|
269
|
+
dry_run: bool,
|
|
270
|
+
offline: bool,
|
|
271
|
+
api_key: str | None,
|
|
272
|
+
fail_on_threshold: bool,
|
|
273
|
+
baseline: Path | None,
|
|
274
|
+
concurrency: int,
|
|
275
|
+
staged: bool,
|
|
276
|
+
changed_since: str | None,
|
|
277
|
+
list_presets: bool,
|
|
278
|
+
cache: bool,
|
|
279
|
+
cache_dir: Path | None,
|
|
280
|
+
exclude_patterns: tuple[str, ...],
|
|
281
|
+
) -> None:
|
|
282
|
+
"""Evaluate documents against quality, safety, or custom evaluation presets."""
|
|
283
|
+
if list_presets:
|
|
284
|
+
click.echo("Available built-in presets:")
|
|
285
|
+
for name in list_builtin_presets():
|
|
286
|
+
loaded_p = load_preset(name)
|
|
287
|
+
click.echo(
|
|
288
|
+
f" • {name:<12} : {loaded_p.title or loaded_p.name} ({loaded_p.description or ''})"
|
|
289
|
+
)
|
|
290
|
+
sys.exit(0)
|
|
291
|
+
|
|
292
|
+
# 1. Load preset configuration early (needed for project config & exclude list)
|
|
293
|
+
preset_cfg: PresetConfig
|
|
294
|
+
if config:
|
|
295
|
+
try:
|
|
296
|
+
preset_cfg = load_preset(str(config))
|
|
297
|
+
except Exception as e:
|
|
298
|
+
err_console.print(f"[bold red]Error loading preset:[/bold red] {e}")
|
|
299
|
+
sys.exit(2)
|
|
300
|
+
elif preset is not None:
|
|
301
|
+
try:
|
|
302
|
+
preset_cfg = load_preset(preset)
|
|
303
|
+
except Exception as e:
|
|
304
|
+
err_console.print(f"[bold red]Error loading preset:[/bold red] {e}")
|
|
305
|
+
sys.exit(2)
|
|
306
|
+
else:
|
|
307
|
+
try:
|
|
308
|
+
auto_cfg, _ = load_project_config()
|
|
309
|
+
if auto_cfg is not None:
|
|
310
|
+
preset_cfg = auto_cfg
|
|
311
|
+
else:
|
|
312
|
+
preset_cfg = load_preset("quality")
|
|
313
|
+
except Exception as e:
|
|
314
|
+
err_console.print(f"[bold red]Error loading discovered config:[/bold red] {e}")
|
|
315
|
+
sys.exit(2)
|
|
316
|
+
|
|
317
|
+
# Git diff resolution if requested
|
|
318
|
+
git_files: list[Path] | None = None
|
|
319
|
+
if staged or changed_since:
|
|
320
|
+
try:
|
|
321
|
+
raw_git_files = get_git_changed_files(staged=staged, since=changed_since)
|
|
322
|
+
git_files = [Path(f).resolve() for f in raw_git_files]
|
|
323
|
+
except RuntimeError as e:
|
|
324
|
+
err_console.print(f"[bold red]Git Error:[/bold red] {e}")
|
|
325
|
+
sys.exit(2)
|
|
326
|
+
|
|
327
|
+
if not files and git_files is None:
|
|
328
|
+
err_console.print("[bold red]Error:[/bold red] No files or file patterns specified.")
|
|
329
|
+
err_console.print("Usage: typesafe-eval [OPTIONS] <FILE_OR_GLOB>...")
|
|
330
|
+
err_console.print("Example: typesafe-eval docs/*.md --preset quality")
|
|
331
|
+
sys.exit(2)
|
|
332
|
+
|
|
333
|
+
# 2. Resolve matched files
|
|
334
|
+
resolved_paths: list[Path] = []
|
|
335
|
+
had_file_matches = False
|
|
336
|
+
|
|
337
|
+
if files:
|
|
338
|
+
for pattern in files:
|
|
339
|
+
if not glob.has_magic(pattern):
|
|
340
|
+
doc_p = Path(pattern)
|
|
341
|
+
if not doc_p.is_file():
|
|
342
|
+
err_console.print(f"[bold red]Error:[/bold red] File not found: {pattern}")
|
|
343
|
+
sys.exit(2)
|
|
344
|
+
if doc_p not in resolved_paths:
|
|
345
|
+
resolved_paths.append(doc_p)
|
|
346
|
+
had_file_matches = True
|
|
347
|
+
else:
|
|
348
|
+
matches = glob.glob(pattern, recursive=True)
|
|
349
|
+
for m in matches:
|
|
350
|
+
doc_p = Path(m)
|
|
351
|
+
if doc_p.is_file():
|
|
352
|
+
had_file_matches = True
|
|
353
|
+
if not is_default_ignored(doc_p) and doc_p not in resolved_paths:
|
|
354
|
+
resolved_paths.append(doc_p)
|
|
355
|
+
|
|
356
|
+
if git_files is not None:
|
|
357
|
+
git_files_set = {f.resolve() for f in git_files}
|
|
358
|
+
resolved_paths = [p for p in resolved_paths if p.resolve() in git_files_set]
|
|
359
|
+
else:
|
|
360
|
+
assert git_files is not None
|
|
361
|
+
had_file_matches = bool(git_files)
|
|
362
|
+
resolved_paths = [
|
|
363
|
+
p
|
|
364
|
+
for p in git_files
|
|
365
|
+
if p.suffix.lower() in DEFAULT_EVAL_EXTENSIONS and not is_default_ignored(p)
|
|
366
|
+
]
|
|
367
|
+
|
|
368
|
+
# 3. Apply exclusion rules from CLI options and preset configuration
|
|
369
|
+
combined_excludes = list(exclude_patterns) + (preset_cfg.exclude or [])
|
|
370
|
+
if combined_excludes:
|
|
371
|
+
resolved_paths = [
|
|
372
|
+
p for p in resolved_paths if not is_path_excluded(p, combined_excludes, root=Path.cwd())
|
|
373
|
+
]
|
|
374
|
+
|
|
375
|
+
if not resolved_paths:
|
|
376
|
+
if staged or changed_since:
|
|
377
|
+
_emit_empty_diff_result(output_format=output_format, out=out)
|
|
378
|
+
sys.exit(0)
|
|
379
|
+
if had_file_matches:
|
|
380
|
+
_emit_empty_diff_result(
|
|
381
|
+
output_format=output_format,
|
|
382
|
+
out=out,
|
|
383
|
+
msg="No files matched evaluation criteria (all matched files were excluded).",
|
|
384
|
+
)
|
|
385
|
+
sys.exit(0)
|
|
386
|
+
err_console.print(
|
|
387
|
+
f"[bold red]Error:[/bold red] No valid files matched the pattern(s): {', '.join(files)}"
|
|
388
|
+
)
|
|
389
|
+
sys.exit(2)
|
|
390
|
+
|
|
391
|
+
# 4. Load baseline if specified
|
|
392
|
+
baseline_lookup = None
|
|
393
|
+
if baseline:
|
|
394
|
+
try:
|
|
395
|
+
baseline_lookup = load_baseline(baseline, expected_preset=preset_cfg.name)
|
|
396
|
+
except FileNotFoundError as e:
|
|
397
|
+
err_console.print(f"[bold red]Error:[/bold red] {e}")
|
|
398
|
+
sys.exit(2)
|
|
399
|
+
except Exception as e:
|
|
400
|
+
err_console.print(f"[bold red]Error loading baseline:[/bold red] {e}")
|
|
401
|
+
sys.exit(2)
|
|
402
|
+
|
|
403
|
+
# 4. Initialize Evaluator
|
|
404
|
+
evaluator = TypeSafeEvaluator(api_key=api_key, enable_cache=cache, cache_dir=cache_dir)
|
|
405
|
+
|
|
406
|
+
# 5. Evaluate documents (concurrent or sequential)
|
|
407
|
+
def _eval_single(target_path: Path) -> tuple[Path, DocumentEvalResult | None, str | None]:
|
|
408
|
+
try:
|
|
409
|
+
res = evaluator.evaluate_document(
|
|
410
|
+
filepath=str(target_path),
|
|
411
|
+
preset=preset_cfg,
|
|
412
|
+
mask_secrets=mask_secrets,
|
|
413
|
+
max_chars=max_chars,
|
|
414
|
+
dry_run=dry_run,
|
|
415
|
+
offline=offline,
|
|
416
|
+
)
|
|
417
|
+
return (target_path, res, None)
|
|
418
|
+
except Exception as e:
|
|
419
|
+
return (target_path, None, str(e))
|
|
420
|
+
|
|
421
|
+
raw_eval_results: list[tuple[Path, DocumentEvalResult | None, str | None]] = []
|
|
422
|
+
if concurrency == 1 or len(resolved_paths) == 1:
|
|
423
|
+
for path in resolved_paths:
|
|
424
|
+
raw_eval_results.append(_eval_single(path))
|
|
425
|
+
else:
|
|
426
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
427
|
+
|
|
428
|
+
with ThreadPoolExecutor(max_workers=min(concurrency, len(resolved_paths))) as executor:
|
|
429
|
+
raw_eval_results = list(executor.map(_eval_single, resolved_paths))
|
|
430
|
+
|
|
431
|
+
results = []
|
|
432
|
+
has_violations = False
|
|
433
|
+
has_errors = False
|
|
434
|
+
|
|
435
|
+
for path, res, err in raw_eval_results:
|
|
436
|
+
if err is not None:
|
|
437
|
+
click.echo(f"{path}: {err}", err=True)
|
|
438
|
+
has_errors = True
|
|
439
|
+
continue
|
|
440
|
+
|
|
441
|
+
assert res is not None
|
|
442
|
+
# Compare against baseline if active
|
|
443
|
+
if baseline_lookup is not None:
|
|
444
|
+
try:
|
|
445
|
+
res, has_reg, warn_msg = compare_document_with_baseline(
|
|
446
|
+
result=res,
|
|
447
|
+
baseline_lookup=baseline_lookup,
|
|
448
|
+
preset=preset_cfg,
|
|
449
|
+
default_max_drop=0.10,
|
|
450
|
+
)
|
|
451
|
+
except ValueError as e:
|
|
452
|
+
err_console.print(f"[bold red]Error comparing baseline:[/bold red] {e}")
|
|
453
|
+
sys.exit(2)
|
|
454
|
+
if warn_msg:
|
|
455
|
+
click.echo(f"Warning: {warn_msg}", err=True)
|
|
456
|
+
if has_reg:
|
|
457
|
+
has_violations = True
|
|
458
|
+
|
|
459
|
+
results.append(res)
|
|
460
|
+
if not res.passed_thresholds and not res.mock:
|
|
461
|
+
has_violations = True
|
|
462
|
+
|
|
463
|
+
# 5. Output handling
|
|
464
|
+
if results or output_format in ("json", "github"):
|
|
465
|
+
if output_format == "table":
|
|
466
|
+
render_table(results, preset_cfg)
|
|
467
|
+
elif output_format == "json":
|
|
468
|
+
json_output = render_json(results)
|
|
469
|
+
click.echo(json_output)
|
|
470
|
+
elif output_format == "markdown":
|
|
471
|
+
md_output = render_markdown(results, preset_cfg)
|
|
472
|
+
click.echo(md_output)
|
|
473
|
+
elif output_format == "github":
|
|
474
|
+
github_output = render_github_annotations(results)
|
|
475
|
+
if github_output:
|
|
476
|
+
click.echo(github_output)
|
|
477
|
+
|
|
478
|
+
# 6. Save to out file if requested
|
|
479
|
+
if out and (results or output_format in ("json", "github")):
|
|
480
|
+
if output_format == "json":
|
|
481
|
+
out.write_text(render_json(results), encoding="utf-8")
|
|
482
|
+
elif output_format == "markdown":
|
|
483
|
+
out.write_text(render_markdown(results, preset_cfg), encoding="utf-8")
|
|
484
|
+
elif output_format == "github":
|
|
485
|
+
out.write_text(render_github_annotations(results), encoding="utf-8")
|
|
486
|
+
else:
|
|
487
|
+
out.write_text(render_markdown(results, preset_cfg), encoding="utf-8")
|
|
488
|
+
err_console.print(f"[green]Report saved successfully to:[/green] {out}")
|
|
489
|
+
|
|
490
|
+
# 7. Exit code resolution (1 takes precedence over 3; dry-run exits 3 on file errors, 0 otherwise)
|
|
491
|
+
if dry_run:
|
|
492
|
+
if has_errors:
|
|
493
|
+
sys.exit(3)
|
|
494
|
+
sys.exit(0)
|
|
495
|
+
elif fail_on_threshold and has_violations:
|
|
496
|
+
sys.exit(1)
|
|
497
|
+
elif has_errors:
|
|
498
|
+
sys.exit(3)
|
|
499
|
+
else:
|
|
500
|
+
sys.exit(0)
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
@main.command(name="validate", context_settings={"help_option_names": ["-h", "--help"]})
|
|
504
|
+
@click.argument("labels_file", required=False, type=str)
|
|
505
|
+
@click.option(
|
|
506
|
+
"--runs",
|
|
507
|
+
type=int,
|
|
508
|
+
default=None,
|
|
509
|
+
help="Number of evaluation runs per document/pair (default: 3, or set in labels file).",
|
|
510
|
+
)
|
|
511
|
+
@click.option(
|
|
512
|
+
"-f",
|
|
513
|
+
"--format",
|
|
514
|
+
"output_format",
|
|
515
|
+
type=click.Choice(["table", "json", "markdown"], case_sensitive=False),
|
|
516
|
+
default="table",
|
|
517
|
+
help="Output presentation format. Default: table.",
|
|
518
|
+
)
|
|
519
|
+
@click.option(
|
|
520
|
+
"-o",
|
|
521
|
+
"--out",
|
|
522
|
+
type=click.Path(dir_okay=False, path_type=Path),
|
|
523
|
+
help="Save report output to specified file path.",
|
|
524
|
+
)
|
|
525
|
+
@click.option(
|
|
526
|
+
"--ablate",
|
|
527
|
+
type=click.Path(exists=True, dir_okay=False, path_type=Path),
|
|
528
|
+
help="Generate 'one section removed' variants from a markdown document split on '## ' headings.",
|
|
529
|
+
)
|
|
530
|
+
@click.option(
|
|
531
|
+
"--ablate-out-dir",
|
|
532
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
533
|
+
help="Output directory for ablated document variants (defaults to document directory).",
|
|
534
|
+
)
|
|
535
|
+
@click.option(
|
|
536
|
+
"--ablate-preset",
|
|
537
|
+
default="design_doc",
|
|
538
|
+
help="Preset name to specify in the generated labels.yaml (default: design_doc).",
|
|
539
|
+
)
|
|
540
|
+
@click.option(
|
|
541
|
+
"--ablate-labels-out",
|
|
542
|
+
type=click.Path(dir_okay=False, path_type=Path),
|
|
543
|
+
help="Write starter labels.yaml to this file path.",
|
|
544
|
+
)
|
|
545
|
+
@click.option(
|
|
546
|
+
"--mask-secrets/--no-mask-secrets",
|
|
547
|
+
"--mask/--no-mask",
|
|
548
|
+
default=True,
|
|
549
|
+
help="Automatically redact detected API keys, credentials, and PII before API call. Default: enabled.",
|
|
550
|
+
)
|
|
551
|
+
@click.option(
|
|
552
|
+
"--dry-run",
|
|
553
|
+
is_flag=True,
|
|
554
|
+
help="Run validation with mock evaluator results.",
|
|
555
|
+
)
|
|
556
|
+
@click.option(
|
|
557
|
+
"--api-key",
|
|
558
|
+
envvar="TYPESAFE_API_KEY",
|
|
559
|
+
help="TypeSafe API key (falls back to TYPESAFE_API_KEY environment variable).",
|
|
560
|
+
)
|
|
561
|
+
def validate_command(
|
|
562
|
+
labels_file: str | None,
|
|
563
|
+
runs: int | None,
|
|
564
|
+
output_format: str,
|
|
565
|
+
out: Path | None,
|
|
566
|
+
ablate: Path | None,
|
|
567
|
+
ablate_out_dir: Path | None,
|
|
568
|
+
ablate_preset: str,
|
|
569
|
+
ablate_labels_out: Path | None,
|
|
570
|
+
dry_run: bool,
|
|
571
|
+
api_key: str | None,
|
|
572
|
+
mask_secrets: bool = True,
|
|
573
|
+
) -> None:
|
|
574
|
+
"""Run validation across fixed test documents using a labels.yaml specification or generate ablation variants."""
|
|
575
|
+
# Handle --ablate helper mode
|
|
576
|
+
if ablate:
|
|
577
|
+
try:
|
|
578
|
+
variants, labels_yaml = generate_ablation_variants(
|
|
579
|
+
doc_path=ablate,
|
|
580
|
+
out_dir=ablate_out_dir,
|
|
581
|
+
preset_name=ablate_preset,
|
|
582
|
+
)
|
|
583
|
+
click.echo(f"Generated {len(variants)} ablation variants:")
|
|
584
|
+
for v in variants:
|
|
585
|
+
click.echo(f" • {v.name}")
|
|
586
|
+
click.echo("\nStarter labels configuration:")
|
|
587
|
+
click.echo(labels_yaml)
|
|
588
|
+
target_labels_out = ablate_labels_out or out
|
|
589
|
+
if target_labels_out:
|
|
590
|
+
target_labels_out.write_text(labels_yaml, encoding="utf-8")
|
|
591
|
+
err_console.print(
|
|
592
|
+
f"[green]Labels configuration saved to:[/green] {target_labels_out}"
|
|
593
|
+
)
|
|
594
|
+
sys.exit(0)
|
|
595
|
+
except Exception as e:
|
|
596
|
+
err_console.print(f"[bold red]Ablation error:[/bold red] {e}")
|
|
597
|
+
sys.exit(2)
|
|
598
|
+
|
|
599
|
+
if not labels_file:
|
|
600
|
+
err_console.print("[bold red]Error:[/bold red] No labels file specified.")
|
|
601
|
+
err_console.print("Usage: typesafe-eval validate <labels.yaml> [OPTIONS]")
|
|
602
|
+
err_console.print(" typesafe-eval validate --ablate <doc.md> [OPTIONS]")
|
|
603
|
+
sys.exit(2)
|
|
604
|
+
|
|
605
|
+
# Load labels file
|
|
606
|
+
try:
|
|
607
|
+
labels_cfg, base_dir = load_labels_file(labels_file)
|
|
608
|
+
except FileNotFoundError as e:
|
|
609
|
+
err_console.print(f"[bold red]Error:[/bold red] {escape(str(e))}")
|
|
610
|
+
sys.exit(2)
|
|
611
|
+
except Exception as e:
|
|
612
|
+
err_console.print(f"[bold red]Error loading labels file:[/bold red] {escape(str(e))}")
|
|
613
|
+
sys.exit(2)
|
|
614
|
+
|
|
615
|
+
# Validate preset and questions
|
|
616
|
+
try:
|
|
617
|
+
preset_cfg = validate_labels_preset(labels_cfg, base_dir)
|
|
618
|
+
except Exception as e:
|
|
619
|
+
err_console.print(f"[bold red]Validation setup error:[/bold red] {e}")
|
|
620
|
+
sys.exit(2)
|
|
621
|
+
|
|
622
|
+
# Initialize evaluator
|
|
623
|
+
evaluator = TypeSafeEvaluator(api_key=api_key)
|
|
624
|
+
|
|
625
|
+
if not dry_run and not evaluator.api_key:
|
|
626
|
+
click.echo(
|
|
627
|
+
"No TypeSafe API key provided. Set the TYPESAFE_API_KEY environment variable "
|
|
628
|
+
"or pass --api-key / specify in configuration.",
|
|
629
|
+
err=True,
|
|
630
|
+
)
|
|
631
|
+
sys.exit(3)
|
|
632
|
+
|
|
633
|
+
# Run validation
|
|
634
|
+
try:
|
|
635
|
+
report, has_runtime_error = run_validation(
|
|
636
|
+
labels_cfg=labels_cfg,
|
|
637
|
+
base_dir=base_dir,
|
|
638
|
+
evaluator=evaluator,
|
|
639
|
+
runs_override=runs,
|
|
640
|
+
dry_run=dry_run,
|
|
641
|
+
preset_cfg=preset_cfg,
|
|
642
|
+
mask_secrets=mask_secrets,
|
|
643
|
+
)
|
|
644
|
+
except Exception as e:
|
|
645
|
+
err_console.print(f"[bold red]Validation setup error:[/bold red] {e}")
|
|
646
|
+
sys.exit(2)
|
|
647
|
+
|
|
648
|
+
# Output unplaced warnings to stderr
|
|
649
|
+
for w in report.unplaced_warnings:
|
|
650
|
+
click.echo(f"Warning: {w}", err=True)
|
|
651
|
+
|
|
652
|
+
# Output handling
|
|
653
|
+
if output_format == "table":
|
|
654
|
+
render_validation_table(report)
|
|
655
|
+
elif output_format == "json":
|
|
656
|
+
json_out = render_validation_json(report)
|
|
657
|
+
click.echo(json_out)
|
|
658
|
+
elif output_format == "markdown":
|
|
659
|
+
md_out = render_validation_markdown(report)
|
|
660
|
+
click.echo(md_out)
|
|
661
|
+
|
|
662
|
+
# Save to out if specified
|
|
663
|
+
if out:
|
|
664
|
+
if output_format == "json":
|
|
665
|
+
out.write_text(render_validation_json(report), encoding="utf-8")
|
|
666
|
+
elif output_format == "markdown":
|
|
667
|
+
out.write_text(render_validation_markdown(report), encoding="utf-8")
|
|
668
|
+
else:
|
|
669
|
+
out.write_text(render_validation_markdown(report), encoding="utf-8")
|
|
670
|
+
err_console.print(f"[green]Report saved successfully to:[/green] {out}")
|
|
671
|
+
|
|
672
|
+
# Exit code resolution (1 over 3 precedence; dry-run exits 3 on file/runtime errors, 0 otherwise)
|
|
673
|
+
if dry_run:
|
|
674
|
+
if has_runtime_error:
|
|
675
|
+
sys.exit(3)
|
|
676
|
+
sys.exit(0)
|
|
677
|
+
elif not report.all_passed:
|
|
678
|
+
sys.exit(1)
|
|
679
|
+
elif has_runtime_error:
|
|
680
|
+
sys.exit(3)
|
|
681
|
+
else:
|
|
682
|
+
sys.exit(0)
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
PRE_COMMIT_SNIPPET = f""" - repo: https://github.com/s-0-a-r/typesafe-eval
|
|
686
|
+
rev: v{__version__}
|
|
687
|
+
hooks:
|
|
688
|
+
- id: typesafe-eval
|
|
689
|
+
args: [--preset, safety]
|
|
690
|
+
"""
|
|
691
|
+
|
|
692
|
+
GITHUB_ACTION_WORKFLOW = f"""name: TypeSafe Evaluation Gate
|
|
693
|
+
|
|
694
|
+
on:
|
|
695
|
+
pull_request:
|
|
696
|
+
paths:
|
|
697
|
+
- '**.md'
|
|
698
|
+
push:
|
|
699
|
+
branches:
|
|
700
|
+
- main
|
|
701
|
+
paths:
|
|
702
|
+
- '**.md'
|
|
703
|
+
|
|
704
|
+
jobs:
|
|
705
|
+
evaluate:
|
|
706
|
+
runs-on: ubuntu-latest
|
|
707
|
+
steps:
|
|
708
|
+
- uses: actions/checkout@v4
|
|
709
|
+
- name: Run typesafe-eval safety gate
|
|
710
|
+
uses: s-0-a-r/typesafe-eval@v{__version__}
|
|
711
|
+
with:
|
|
712
|
+
preset: safety
|
|
713
|
+
env:
|
|
714
|
+
TYPESAFE_API_KEY: ${{{{ secrets.TYPESAFE_API_KEY }}}}
|
|
715
|
+
"""
|
|
716
|
+
|
|
717
|
+
CLAUDE_HOOK_CONFIG = """{
|
|
718
|
+
"hooks": {
|
|
719
|
+
"PostToolUse": [
|
|
720
|
+
{
|
|
721
|
+
"matcher": "Edit|Write|MultiEdit",
|
|
722
|
+
"command": "python3 hooks/claude_safety_hook.py"
|
|
723
|
+
}
|
|
724
|
+
]
|
|
725
|
+
}
|
|
726
|
+
}
|
|
727
|
+
"""
|
|
728
|
+
|
|
729
|
+
|
|
730
|
+
@main.command(name="init", context_settings={"help_option_names": ["-h", "--help"]})
|
|
731
|
+
@click.option(
|
|
732
|
+
"--pre-commit", "opt_pre_commit", is_flag=True, help="Configure .pre-commit-config.yaml hook."
|
|
733
|
+
)
|
|
734
|
+
@click.option(
|
|
735
|
+
"--claude-code",
|
|
736
|
+
"opt_claude_code",
|
|
737
|
+
is_flag=True,
|
|
738
|
+
help="Configure Claude Code safety hook (hooks/hooks.json).",
|
|
739
|
+
)
|
|
740
|
+
@click.option(
|
|
741
|
+
"--github-action",
|
|
742
|
+
"opt_github_action",
|
|
743
|
+
is_flag=True,
|
|
744
|
+
help="Generate .github/workflows/typesafe-eval.yml.",
|
|
745
|
+
)
|
|
746
|
+
@click.option("--all", "opt_all", is_flag=True, help="Set up all agent and CI integrations.")
|
|
747
|
+
def init(
|
|
748
|
+
opt_pre_commit: bool,
|
|
749
|
+
opt_claude_code: bool,
|
|
750
|
+
opt_github_action: bool,
|
|
751
|
+
opt_all: bool,
|
|
752
|
+
) -> None:
|
|
753
|
+
"""Scaffolds agent integrations and CI workflows in the current repository."""
|
|
754
|
+
if opt_all or not (opt_pre_commit or opt_claude_code or opt_github_action):
|
|
755
|
+
opt_pre_commit = True
|
|
756
|
+
opt_claude_code = True
|
|
757
|
+
if opt_all:
|
|
758
|
+
opt_github_action = True
|
|
759
|
+
|
|
760
|
+
configured = []
|
|
761
|
+
|
|
762
|
+
# 1. Pre-commit
|
|
763
|
+
if opt_pre_commit:
|
|
764
|
+
pc_path = Path(".pre-commit-config.yaml")
|
|
765
|
+
if pc_path.exists():
|
|
766
|
+
content = pc_path.read_text(encoding="utf-8")
|
|
767
|
+
if "typesafe-eval" not in content:
|
|
768
|
+
if "repos:" in content:
|
|
769
|
+
content = content.replace("repos:\n", f"repos:\n{PRE_COMMIT_SNIPPET}")
|
|
770
|
+
else:
|
|
771
|
+
content += f"\nrepos:\n{PRE_COMMIT_SNIPPET}"
|
|
772
|
+
pc_path.write_text(content, encoding="utf-8")
|
|
773
|
+
configured.append(".pre-commit-config.yaml (updated)")
|
|
774
|
+
else:
|
|
775
|
+
configured.append(".pre-commit-config.yaml (already configured)")
|
|
776
|
+
else:
|
|
777
|
+
pc_path.write_text(f"repos:\n{PRE_COMMIT_SNIPPET}", encoding="utf-8")
|
|
778
|
+
configured.append(".pre-commit-config.yaml (created)")
|
|
779
|
+
|
|
780
|
+
# 2. Claude Code Hook
|
|
781
|
+
if opt_claude_code:
|
|
782
|
+
hooks_dir = Path("hooks")
|
|
783
|
+
hooks_dir.mkdir(exist_ok=True)
|
|
784
|
+
hh_path = hooks_dir / "hooks.json"
|
|
785
|
+
if not hh_path.exists():
|
|
786
|
+
hh_path.write_text(CLAUDE_HOOK_CONFIG, encoding="utf-8")
|
|
787
|
+
configured.append("hooks/hooks.json (created)")
|
|
788
|
+
else:
|
|
789
|
+
configured.append("hooks/hooks.json (already exists)")
|
|
790
|
+
|
|
791
|
+
# 3. GitHub Action Workflow
|
|
792
|
+
if opt_github_action:
|
|
793
|
+
wf_dir = Path(".github/workflows")
|
|
794
|
+
wf_dir.mkdir(parents=True, exist_ok=True)
|
|
795
|
+
wf_path = wf_dir / "typesafe-eval.yml"
|
|
796
|
+
if not wf_path.exists():
|
|
797
|
+
wf_path.write_text(GITHUB_ACTION_WORKFLOW, encoding="utf-8")
|
|
798
|
+
configured.append(".github/workflows/typesafe-eval.yml (created)")
|
|
799
|
+
else:
|
|
800
|
+
configured.append(".github/workflows/typesafe-eval.yml (already exists)")
|
|
801
|
+
|
|
802
|
+
for item in configured:
|
|
803
|
+
click.echo(f"✓ {item}")
|
|
804
|
+
sys.exit(0)
|
|
805
|
+
|
|
806
|
+
|
|
807
|
+
@main.command(name="schema", context_settings={"help_option_names": ["-h", "--help"]})
|
|
808
|
+
@click.option(
|
|
809
|
+
"-t",
|
|
810
|
+
"--type",
|
|
811
|
+
"schema_type",
|
|
812
|
+
type=click.Choice(["preset", "labels"], case_sensitive=False),
|
|
813
|
+
default="preset",
|
|
814
|
+
help="Schema target type ('preset' for evaluation config YAML, 'labels' for validation labels YAML). Default: preset.",
|
|
815
|
+
)
|
|
816
|
+
@click.option(
|
|
817
|
+
"-o",
|
|
818
|
+
"--out",
|
|
819
|
+
type=click.Path(dir_okay=False, path_type=Path),
|
|
820
|
+
help="Save JSON schema output to specified file path.",
|
|
821
|
+
)
|
|
822
|
+
@click.option(
|
|
823
|
+
"--indent",
|
|
824
|
+
type=int,
|
|
825
|
+
default=2,
|
|
826
|
+
help="Indentation spaces for JSON formatting. Default: 2.",
|
|
827
|
+
)
|
|
828
|
+
def schema(schema_type: str, out: Path | None, indent: int) -> None:
|
|
829
|
+
"""Outputs JSON Schema for configuration and presets (enables IDE autocomplete)."""
|
|
830
|
+
import json
|
|
831
|
+
|
|
832
|
+
from typesafe_eval.models import PresetConfig
|
|
833
|
+
from typesafe_eval.validator import ValidationLabelsConfig
|
|
834
|
+
|
|
835
|
+
if schema_type.lower() == "labels":
|
|
836
|
+
schema_dict = ValidationLabelsConfig.model_json_schema()
|
|
837
|
+
schema_dict["title"] = "TypeSafeEvalValidationLabels"
|
|
838
|
+
schema_dict["description"] = (
|
|
839
|
+
"JSON Schema for typesafe-eval ground-truth labels specification (labels.yaml)."
|
|
840
|
+
)
|
|
841
|
+
else:
|
|
842
|
+
schema_dict = PresetConfig.model_json_schema()
|
|
843
|
+
schema_dict["title"] = "TypeSafeEvalPresetConfig"
|
|
844
|
+
schema_dict["description"] = (
|
|
845
|
+
"JSON Schema for typesafe-eval evaluation preset and configuration YAML files."
|
|
846
|
+
)
|
|
847
|
+
|
|
848
|
+
json_str = json.dumps(schema_dict, indent=indent)
|
|
849
|
+
if out:
|
|
850
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
851
|
+
out.write_text(json_str + "\n", encoding="utf-8")
|
|
852
|
+
click.echo(f"✓ JSON Schema saved to {out}")
|
|
853
|
+
else:
|
|
854
|
+
click.echo(json_str)
|
|
855
|
+
|
|
856
|
+
|
|
857
|
+
@main.group(name="cache")
|
|
858
|
+
def cache_group() -> None:
|
|
859
|
+
"""Manage local evaluation result cache."""
|
|
860
|
+
pass
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
@cache_group.command(name="clear")
|
|
864
|
+
@click.option(
|
|
865
|
+
"--cache-dir",
|
|
866
|
+
type=click.Path(file_okay=False, path_type=Path),
|
|
867
|
+
default=None,
|
|
868
|
+
help="Custom directory for caching evaluation results.",
|
|
869
|
+
)
|
|
870
|
+
def cache_clear_command(cache_dir: Path | None) -> None:
|
|
871
|
+
"""Clear all cached evaluation results."""
|
|
872
|
+
from typesafe_eval.cache import EvaluationCache
|
|
873
|
+
|
|
874
|
+
c = EvaluationCache(cache_dir=cache_dir)
|
|
875
|
+
count = c.clear()
|
|
876
|
+
click.echo(f"Cleared {count} cached evaluation result(s).")
|
|
877
|
+
|
|
878
|
+
|
|
879
|
+
if __name__ == "__main__":
|
|
880
|
+
main()
|