alignmenter 0.3.0__tar.gz → 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {alignmenter-0.3.0/src/alignmenter.egg-info → alignmenter-0.3.1}/PKG-INFO +1 -1
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/_version.py +1 -1
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/release_cli.py +22 -2
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/durable.py +3 -1
- alignmenter-0.3.1/src/alignmenter/reporting/github_comment.py +140 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1/src/alignmenter.egg-info}/PKG-INFO +1 -1
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/SOURCES.txt +2 -0
- alignmenter-0.3.1/tests/test_github_comment.py +116 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/LICENSE +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/MANIFEST.in +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/README.md +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/demo_config.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/judges/safety_prompt.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/persona/default.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/run-grounded.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/run.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/safety_keywords.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/README.md +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/demo_conversations.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/grounded_demo.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/wendys_twitter.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/pyproject.toml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/setup.cfg +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/analyze.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/bounds.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/diagnose.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/generate.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/label.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/optimize.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/sampling.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/validate.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/cli.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/config.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/demo_config.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/judges/safety_prompt.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/persona/default.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/run-grounded.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/run.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/safety_keywords.yaml +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/datasets/demo_conversations.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/datasets/grounded_demo.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/custom.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/evidence.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/faithfulness.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/grounding.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/metrics.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/examples/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/examples/resource_task.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/archive.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/artifacts.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/comparison.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/evaluation.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/gates.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/leases.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/legacy.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/recovery.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/review.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/suite.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/authenticity_judge.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/prompts.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/anthropic.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/base.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/callable.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/classifiers.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/durable_judge.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/embeddings.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/judges.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/local.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/openai.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/html.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/json_out.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/run_config.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/runner.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/evaluation.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/execution.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/gates.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/metrics.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/review.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/scoring.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/suite.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/authenticity.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/faithfulness.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/grounding.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/safety.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/stability.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/bootstrap_dataset.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/calibrate_persona.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/run_openai_demo.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/sanitize_dataset.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/sdk.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/evaluations.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/reviews.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/runs.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/io.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/optional.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/tokens.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/yaml.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/dependency_links.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/entry_points.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/requires.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/top_level.txt +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/__init__.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/conftest.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_evaluation_judge.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_evaluation_worker.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_recovery_target.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_recovery_worker.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_run_worker.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/mini_cli_dataset.jsonl +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_authenticity_judge.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_builtin_evaluations.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_calibrate_persona.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_capture_recovery.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_errors.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_grounded.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_helpers.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_import.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_init.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_run_config.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_config.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_durable_evaluations.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_durable_execution.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_faithfulness.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_grounding.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_html_report.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_judge_providers.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_offline_safety.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_persona_gpt.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_provider_local.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_provider_openai.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_providers.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_release_workflow.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_review_workflow.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_config_grounded.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_config_loader.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_openai_demo.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_runner.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_sampling.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_scorers.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_smoke.py +0 -0
- {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_suite_archive.py +0 -0
|
@@ -22,6 +22,18 @@ from alignmenter.schemas.gates import GatePolicy
|
|
|
22
22
|
EXIT_CODES = {"pass": 0, "fail": 2, "inconclusive": 3}
|
|
23
23
|
|
|
24
24
|
|
|
25
|
+
def _exit_code(decision: str, *, allow_inconclusive: bool = False) -> int:
|
|
26
|
+
"""Map a decision to a process exit code.
|
|
27
|
+
|
|
28
|
+
With ``allow_inconclusive`` an inconclusive decision (e.g. an unreviewed
|
|
29
|
+
``draft`` spec that met every applicable criterion) exits 0 so an automated
|
|
30
|
+
gate is not blocked; a genuine ``fail`` still exits non-zero.
|
|
31
|
+
"""
|
|
32
|
+
if allow_inconclusive and decision == "inconclusive":
|
|
33
|
+
return 0
|
|
34
|
+
return EXIT_CODES[decision]
|
|
35
|
+
|
|
36
|
+
|
|
25
37
|
def register_release_commands(app):
|
|
26
38
|
@app.command("archive-export")
|
|
27
39
|
def archive_export(
|
|
@@ -67,6 +79,9 @@ def register_release_commands(app):
|
|
|
67
79
|
suite: Path = typer.Argument(..., exists=True, dir_okay=False),
|
|
68
80
|
out: Path = typer.Option(Path("reports"), "--out"),
|
|
69
81
|
resume: Path | None = typer.Option(None, "--resume", exists=True, file_okay=False),
|
|
82
|
+
allow_inconclusive: bool = typer.Option(
|
|
83
|
+
False, "--allow-inconclusive",
|
|
84
|
+
help="Exit 0 on an inconclusive decision (a fail still exits non-zero)."),
|
|
70
85
|
):
|
|
71
86
|
"""Capture, evaluate, compare, and write CI artifacts under a frozen suite config."""
|
|
72
87
|
try:
|
|
@@ -74,7 +89,9 @@ def register_release_commands(app):
|
|
|
74
89
|
except Exception as exc:
|
|
75
90
|
raise typer.BadParameter(str(exc)) from exc
|
|
76
91
|
typer.echo(json.dumps(result, indent=2))
|
|
77
|
-
|
|
92
|
+
if allow_inconclusive and result["decision"] == "inconclusive":
|
|
93
|
+
typer.echo("Inconclusive tolerated (--allow-inconclusive): exiting 0.")
|
|
94
|
+
raise typer.Exit(_exit_code(result["decision"], allow_inconclusive=allow_inconclusive))
|
|
78
95
|
|
|
79
96
|
@app.command("review-export")
|
|
80
97
|
def review_export(
|
|
@@ -137,6 +154,9 @@ def register_release_commands(app):
|
|
|
137
154
|
baseline: Path | None = typer.Option(None, "--baseline", exists=True, file_okay=False),
|
|
138
155
|
baseline_id: UUID | None = typer.Option(None, "--baseline-id"),
|
|
139
156
|
force: bool = typer.Option(False, "--force"),
|
|
157
|
+
allow_inconclusive: bool = typer.Option(
|
|
158
|
+
False, "--allow-inconclusive",
|
|
159
|
+
help="Exit 0 on an inconclusive decision (a fail still exits non-zero)."),
|
|
140
160
|
):
|
|
141
161
|
"""Check saved results and export CI artifacts without invoking any provider."""
|
|
142
162
|
try:
|
|
@@ -147,7 +167,7 @@ def register_release_commands(app):
|
|
|
147
167
|
raise typer.BadParameter(str(exc)) from exc
|
|
148
168
|
decision = report["gate_report"]["decision"]
|
|
149
169
|
typer.echo(f"Decision: {decision}\nArtifacts: {out.resolve()}")
|
|
150
|
-
raise typer.Exit(
|
|
170
|
+
raise typer.Exit(_exit_code(decision, allow_inconclusive=allow_inconclusive))
|
|
151
171
|
|
|
152
172
|
@app.command("compare")
|
|
153
173
|
def compare(
|
|
@@ -12,6 +12,7 @@ from xml.etree import ElementTree as ET
|
|
|
12
12
|
|
|
13
13
|
from alignmenter.execution.evaluation import evaluation_summary
|
|
14
14
|
from alignmenter.execution.gates import gate_report
|
|
15
|
+
from alignmenter.reporting.github_comment import render_github_comment
|
|
15
16
|
|
|
16
17
|
|
|
17
18
|
def _escape(value):
|
|
@@ -167,5 +168,6 @@ def export_evaluation(run_dir, out_dir, *, evaluation_id=None, policy=None, comp
|
|
|
167
168
|
report["comparison"] = comparison
|
|
168
169
|
report["review"] = qualification_report(run_dir, report["evaluation_id"])
|
|
169
170
|
write_artifacts(out_dir, {"evaluation.json": _pretty(report) + "\n", "index.html": render_html(report),
|
|
170
|
-
"junit.xml": render_junit(report), "summary.md": render_markdown(report)
|
|
171
|
+
"junit.xml": render_junit(report), "summary.md": render_markdown(report),
|
|
172
|
+
"comment.md": render_github_comment(report)}, force=force)
|
|
171
173
|
return report
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""Sticky GitHub pull-request comment over one saved decision; never executes a scorer.
|
|
2
|
+
|
|
3
|
+
Renders the same ``report`` structure the other durable reporters consume
|
|
4
|
+
(:mod:`alignmenter.reporting.durable`) into Markdown tuned for a PR comment: a
|
|
5
|
+
verdict badge, blocking issues first, a gate table, metrics (with baseline
|
|
6
|
+
deltas when a comparison is present), and collapsed breakdowns. The leading
|
|
7
|
+
marker lets a CI step find and update one sticky comment instead of appending.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
MARKER = "<!-- alignmenter:report -->"
|
|
13
|
+
|
|
14
|
+
_DECISION_BADGE = {"pass": "✅ Pass", "fail": "❌ Fail", "inconclusive": "⚠️ Inconclusive"}
|
|
15
|
+
_CHECK_ICON = {"pass": "✅", "fail": "❌", "inconclusive": "➖"}
|
|
16
|
+
_OPERATOR = {"at_least": "≥", "at_most": "≤"}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _cell(value):
|
|
20
|
+
"""Escape a value for a single Markdown table cell."""
|
|
21
|
+
if value is None:
|
|
22
|
+
return "—"
|
|
23
|
+
return (str(value).replace("\\", "\\\\").replace("|", "\\|")
|
|
24
|
+
.replace("\n", " ").replace("<", "<").replace(">", ">"))
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _num(value, digits=3):
|
|
28
|
+
if value is None:
|
|
29
|
+
return "—"
|
|
30
|
+
if isinstance(value, bool):
|
|
31
|
+
return str(value)
|
|
32
|
+
if isinstance(value, float):
|
|
33
|
+
return f"{value:.{digits}f}".rstrip("0").rstrip(".") or "0"
|
|
34
|
+
return str(value)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _signed(value, digits=3):
|
|
38
|
+
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
39
|
+
return ("+" if value > 0 else "") + _num(value, digits)
|
|
40
|
+
return _num(value, digits)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _icon(decision):
|
|
44
|
+
return _CHECK_ICON.get(decision, _cell(decision))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _gate_detail(check):
|
|
48
|
+
metric = check.get("metric")
|
|
49
|
+
if metric is None:
|
|
50
|
+
return _cell(check.get("reason", ""))
|
|
51
|
+
operator = _OPERATOR.get(check.get("operator"), check.get("operator") or "")
|
|
52
|
+
return f"`{_cell(metric)}` {operator} {_num(check.get('threshold'))} · got **{_num(check.get('value'))}**"
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _blocking_lines(report):
|
|
56
|
+
lines = []
|
|
57
|
+
for check in report["gate_report"].get("checks", []):
|
|
58
|
+
# Skip the structural "evaluation"/"comparison" umbrella checks (no metric);
|
|
59
|
+
# the ❌ badge and the violated cases below already convey those.
|
|
60
|
+
if check["decision"] == "fail" and check.get("metric") is not None:
|
|
61
|
+
lines.append(f"- **{_cell(check['id'])}** — {_gate_detail(check)}")
|
|
62
|
+
inputs = {item["key"]: item for item in report.get("inputs", [])}
|
|
63
|
+
violated = [(inputs.get(r["input_key"]), r) for r in report.get("results", [])
|
|
64
|
+
if r.get("status") == "violated"]
|
|
65
|
+
for item, result in violated[:5]:
|
|
66
|
+
item = item or {}
|
|
67
|
+
label = _cell(item.get("case_id") or item.get("session_id") or "case")
|
|
68
|
+
lines.append(f"- case **{label}** / {_cell(item.get('criterion_id'))}: "
|
|
69
|
+
f"{_cell(result.get('reason') or 'violated')}")
|
|
70
|
+
if len(violated) > 5:
|
|
71
|
+
lines.append(f"- …and {len(violated) - 5} more violated case(s)")
|
|
72
|
+
return lines
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _gate_table(report):
|
|
76
|
+
checks = report["gate_report"].get("checks", [])
|
|
77
|
+
if not checks:
|
|
78
|
+
return []
|
|
79
|
+
rows = ["| Gate | Result | Detail |", "| --- | :---: | --- |"]
|
|
80
|
+
rows += [f"| {_cell(c['id'])} | {_icon(c['decision'])} | {_gate_detail(c)} |" for c in checks]
|
|
81
|
+
return ["#### Gates", *rows, ""]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _metrics_table(report):
|
|
85
|
+
comparison = report.get("comparison")
|
|
86
|
+
if comparison and comparison.get("metrics"):
|
|
87
|
+
rows = ["| Metric | Baseline | Candidate | Δ | Note |",
|
|
88
|
+
"| --- | ---: | ---: | ---: | :---: |"]
|
|
89
|
+
for name, metric in comparison["metrics"].items():
|
|
90
|
+
note = "⚠️ unavailable" if metric.get("unavailable") else ""
|
|
91
|
+
rows.append(f"| `{_cell(name)}` | {_num(metric.get('baseline'))} | "
|
|
92
|
+
f"{_num(metric.get('candidate'))} | {_signed(metric.get('delta'))} | {note} |")
|
|
93
|
+
return ["#### Metrics (vs baseline)", *rows, ""]
|
|
94
|
+
metrics = report.get("metrics") or {}
|
|
95
|
+
if not metrics:
|
|
96
|
+
return []
|
|
97
|
+
rows = ["| Metric | Value | Denominator |", "| --- | ---: | ---: |"]
|
|
98
|
+
rows += [f"| `{_cell(name)}` | {_num(m.get('value'))} | {_num(m.get('denominator'))} |"
|
|
99
|
+
for name, m in metrics.items()]
|
|
100
|
+
return ["#### Metrics", *rows, ""]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _details(summary, body):
|
|
104
|
+
return [f"<details><summary>{_cell(summary)}</summary>", "", body, "", "</details>", ""]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _breakdown(report):
|
|
108
|
+
lines = []
|
|
109
|
+
criteria = report.get("criteria") or {}
|
|
110
|
+
if criteria:
|
|
111
|
+
rows = ["| Criterion | met | violated | n/a | decision |",
|
|
112
|
+
"| --- | ---: | ---: | ---: | :---: |"]
|
|
113
|
+
for cid, summary in criteria.items():
|
|
114
|
+
counts = summary.get("counts", {})
|
|
115
|
+
rows.append(f"| `{_cell(cid)}` | {counts.get('met', 0)} | {counts.get('violated', 0)} | "
|
|
116
|
+
f"{counts.get('not_applicable', 0)} | {_icon(summary.get('decision'))} |")
|
|
117
|
+
lines += _details("Per-criterion breakdown", "\n".join(rows))
|
|
118
|
+
counts = report.get("counts") or {}
|
|
119
|
+
if counts:
|
|
120
|
+
lines += _details("Outcome counts", ", ".join(f"{k}: {v}" for k, v in counts.items()))
|
|
121
|
+
return lines
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def render_github_comment(report, *, title="Alignmenter"):
|
|
125
|
+
"""Return a sticky PR-comment Markdown body for one saved evaluation ``report``."""
|
|
126
|
+
decision = report["gate_report"]["decision"]
|
|
127
|
+
spec = report["spec"]
|
|
128
|
+
header = (f"`{_cell(spec['id'])} @ {_cell(spec['revision'])}` · "
|
|
129
|
+
f"qualification `{_cell(spec['qualification'])}` · "
|
|
130
|
+
f"{report['judged']}/{report['applicable']} applicable criteria evaluated")
|
|
131
|
+
if report.get("unavailable"):
|
|
132
|
+
header += f" · {report['unavailable']} unavailable"
|
|
133
|
+
lines = [MARKER, f"### {_DECISION_BADGE.get(decision, _cell(decision))} — {_cell(title)}", header, ""]
|
|
134
|
+
blocking = _blocking_lines(report)
|
|
135
|
+
if blocking:
|
|
136
|
+
lines += ["#### ❌ Blocking", *blocking, ""]
|
|
137
|
+
lines += _gate_table(report)
|
|
138
|
+
lines += _metrics_table(report)
|
|
139
|
+
lines += _breakdown(report)
|
|
140
|
+
return "\n".join(lines).rstrip() + "\n"
|
|
@@ -77,6 +77,7 @@ src/alignmenter/providers/local.py
|
|
|
77
77
|
src/alignmenter/providers/openai.py
|
|
78
78
|
src/alignmenter/reporting/__init__.py
|
|
79
79
|
src/alignmenter/reporting/durable.py
|
|
80
|
+
src/alignmenter/reporting/github_comment.py
|
|
80
81
|
src/alignmenter/reporting/html.py
|
|
81
82
|
src/alignmenter/reporting/json_out.py
|
|
82
83
|
src/alignmenter/schemas/__init__.py
|
|
@@ -123,6 +124,7 @@ tests/test_config.py
|
|
|
123
124
|
tests/test_durable_evaluations.py
|
|
124
125
|
tests/test_durable_execution.py
|
|
125
126
|
tests/test_faithfulness.py
|
|
127
|
+
tests/test_github_comment.py
|
|
126
128
|
tests/test_grounding.py
|
|
127
129
|
tests/test_html_report.py
|
|
128
130
|
tests/test_judge_providers.py
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""GitHub PR-comment reporter and the --allow-inconclusive CI exit flag."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typer.testing import CliRunner
|
|
6
|
+
|
|
7
|
+
from alignmenter.cli import app
|
|
8
|
+
from alignmenter.execution.evaluation import evaluate_saved
|
|
9
|
+
from alignmenter.release_cli import _exit_code
|
|
10
|
+
from alignmenter.reporting.durable import export_evaluation
|
|
11
|
+
from alignmenter.reporting.github_comment import MARKER, render_github_comment
|
|
12
|
+
from alignmenter.schemas.evaluation import JudgeBudget
|
|
13
|
+
|
|
14
|
+
from .test_durable_evaluations import Judge, captured, spec
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def make_report(*, decision="pass", checks=None, metrics=None, comparison=None,
|
|
18
|
+
criteria=None, counts=None, inputs=None, results=None,
|
|
19
|
+
judged=2, applicable=2, unavailable=0, qualification="reviewed"):
|
|
20
|
+
return {
|
|
21
|
+
"spec": {"id": "avercare", "revision": "v1", "qualification": qualification},
|
|
22
|
+
"judged": judged, "applicable": applicable, "unavailable": unavailable,
|
|
23
|
+
"metrics": metrics or {}, "comparison": comparison,
|
|
24
|
+
"criteria": criteria or {}, "counts": counts or {},
|
|
25
|
+
"inputs": inputs or [], "results": results or [],
|
|
26
|
+
"gate_report": {"decision": decision, "checks": checks or [
|
|
27
|
+
{"id": "evaluation", "decision": decision, "reason": "All required outcomes."}]},
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_pass_comment_has_marker_badge_and_header():
|
|
32
|
+
body = render_github_comment(make_report(decision="pass"))
|
|
33
|
+
assert body.splitlines()[0] == MARKER # first line so a sticky-comment step can match it
|
|
34
|
+
assert "✅ Pass" in body
|
|
35
|
+
assert "`avercare @ v1`" in body and "2/2 applicable criteria" in body
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_failing_gate_and_violated_case_surface_in_blocking_section():
|
|
39
|
+
report = make_report(
|
|
40
|
+
decision="fail",
|
|
41
|
+
checks=[
|
|
42
|
+
{"id": "evaluation", "decision": "fail", "reason": "A criterion was violated."},
|
|
43
|
+
{"id": "dangerous_zero", "decision": "fail", "metric": "faithfulness.dangerous_answers",
|
|
44
|
+
"operator": "at_most", "threshold": 0, "value": 2},
|
|
45
|
+
],
|
|
46
|
+
inputs=[{"key": "k1", "case_id": "hydration-01", "criterion_id": "faithfulness"}],
|
|
47
|
+
results=[{"input_key": "k1", "status": "violated", "reason": "Unsupported dosage claim."}],
|
|
48
|
+
)
|
|
49
|
+
body = render_github_comment(report)
|
|
50
|
+
blocking = body.split("#### ❌ Blocking", 1)[1]
|
|
51
|
+
assert "❌ Fail" in body
|
|
52
|
+
assert "**dangerous_zero**" in blocking and "≤ 0" in blocking and "got **2**" in blocking
|
|
53
|
+
assert "hydration-01" in blocking and "Unsupported dosage claim." in blocking
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def test_inconclusive_badge_and_no_blocking_section():
|
|
57
|
+
body = render_github_comment(make_report(decision="inconclusive", qualification="draft"))
|
|
58
|
+
assert "⚠️ Inconclusive" in body
|
|
59
|
+
assert "#### ❌ Blocking" not in body
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_comparison_renders_baseline_delta_columns():
|
|
63
|
+
report = make_report(
|
|
64
|
+
decision="pass",
|
|
65
|
+
comparison={"metrics": {
|
|
66
|
+
"evaluation.met_rate": {"baseline": 0.8, "candidate": 0.9, "delta": 0.1, "unavailable": False},
|
|
67
|
+
"grounding.citation_resolution": {"baseline": None, "candidate": None, "delta": None, "unavailable": True},
|
|
68
|
+
}},
|
|
69
|
+
)
|
|
70
|
+
body = render_github_comment(report)
|
|
71
|
+
assert "Metrics (vs baseline)" in body
|
|
72
|
+
assert "+0.1" in body # signed delta
|
|
73
|
+
assert "⚠️ unavailable" in body # unavailable metric flagged, not dropped
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def test_table_cells_are_escaped_and_do_not_break_the_marker():
|
|
77
|
+
report = make_report(
|
|
78
|
+
decision="fail",
|
|
79
|
+
checks=[{"id": "evaluation", "decision": "fail", "reason": "pipe | and\nnewline <b>"}],
|
|
80
|
+
inputs=[{"key": "k1", "case_id": "a|b", "criterion_id": "c"}],
|
|
81
|
+
results=[{"input_key": "k1", "status": "violated", "reason": "x | y\nz"}],
|
|
82
|
+
)
|
|
83
|
+
body = render_github_comment(report)
|
|
84
|
+
assert body.splitlines()[0] == MARKER
|
|
85
|
+
# no raw pipe/newline leaks into a table cell that would break rendering
|
|
86
|
+
assert "a\\|b" in body and "x \\| y z" in body
|
|
87
|
+
assert "<b>" not in body
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def test_exit_code_mapping_and_allow_inconclusive():
|
|
91
|
+
assert _exit_code("pass") == 0
|
|
92
|
+
assert _exit_code("fail") == 2
|
|
93
|
+
assert _exit_code("inconclusive") == 3
|
|
94
|
+
assert _exit_code("inconclusive", allow_inconclusive=True) == 0
|
|
95
|
+
assert _exit_code("fail", allow_inconclusive=True) == 2 # a real failure still blocks
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_export_writes_comment_alongside_other_artifacts(tmp_path):
|
|
99
|
+
run_dir = captured(tmp_path)
|
|
100
|
+
evaluate_saved(run_dir, spec(), Judge(), budget=JudgeBudget(max_calls=10)) # reviewed + met => pass
|
|
101
|
+
out = tmp_path / "reports"
|
|
102
|
+
export_evaluation(run_dir, out)
|
|
103
|
+
for name in ("evaluation.json", "index.html", "junit.xml", "summary.md", "comment.md"):
|
|
104
|
+
assert (out / name).exists(), name
|
|
105
|
+
comment = (out / "comment.md").read_text()
|
|
106
|
+
assert comment.startswith(MARKER) and "✅ Pass" in comment
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def test_check_cli_allow_inconclusive_flips_exit_but_not_failure(tmp_path):
|
|
110
|
+
run_dir = captured(tmp_path)
|
|
111
|
+
evaluate_saved(run_dir, spec(qualification="draft"), Judge(), budget=JudgeBudget(max_calls=10)) # met, but draft => inconclusive
|
|
112
|
+
default = CliRunner().invoke(app, ["check", str(run_dir), "--out", str(tmp_path / "r1")])
|
|
113
|
+
assert default.exit_code == 3, default.output
|
|
114
|
+
allowed = CliRunner().invoke(app, ["check", str(run_dir), "--out", str(tmp_path / "r2"), "--allow-inconclusive"])
|
|
115
|
+
assert allowed.exit_code == 0, allowed.output
|
|
116
|
+
assert (tmp_path / "r2" / "comment.md").exists()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/judges/safety_prompt.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/datasets/demo_conversations.jsonl
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|