alignmenter 0.3.0__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (153) hide show
  1. {alignmenter-0.3.0/src/alignmenter.egg-info → alignmenter-0.3.1}/PKG-INFO +1 -1
  2. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/_version.py +1 -1
  3. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/release_cli.py +22 -2
  4. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/durable.py +3 -1
  5. alignmenter-0.3.1/src/alignmenter/reporting/github_comment.py +140 -0
  6. {alignmenter-0.3.0 → alignmenter-0.3.1/src/alignmenter.egg-info}/PKG-INFO +1 -1
  7. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/SOURCES.txt +2 -0
  8. alignmenter-0.3.1/tests/test_github_comment.py +116 -0
  9. {alignmenter-0.3.0 → alignmenter-0.3.1}/LICENSE +0 -0
  10. {alignmenter-0.3.0 → alignmenter-0.3.1}/MANIFEST.in +0 -0
  11. {alignmenter-0.3.0 → alignmenter-0.3.1}/README.md +0 -0
  12. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/demo_config.yaml +0 -0
  13. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/judges/safety_prompt.txt +0 -0
  14. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/persona/default.yaml +0 -0
  15. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/run-grounded.yaml +0 -0
  16. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/run.yaml +0 -0
  17. {alignmenter-0.3.0 → alignmenter-0.3.1}/configs/safety_keywords.yaml +0 -0
  18. {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/README.md +0 -0
  19. {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/demo_conversations.jsonl +0 -0
  20. {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/grounded_demo.jsonl +0 -0
  21. {alignmenter-0.3.0 → alignmenter-0.3.1}/datasets/wendys_twitter.jsonl +0 -0
  22. {alignmenter-0.3.0 → alignmenter-0.3.1}/pyproject.toml +0 -0
  23. {alignmenter-0.3.0 → alignmenter-0.3.1}/setup.cfg +0 -0
  24. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/__init__.py +0 -0
  25. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/__init__.py +0 -0
  26. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/analyze.py +0 -0
  27. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/bounds.py +0 -0
  28. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/diagnose.py +0 -0
  29. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/generate.py +0 -0
  30. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/label.py +0 -0
  31. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/optimize.py +0 -0
  32. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/sampling.py +0 -0
  33. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/calibration/validate.py +0 -0
  34. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/cli.py +0 -0
  35. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/config.py +0 -0
  36. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/demo_config.yaml +0 -0
  37. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/judges/safety_prompt.txt +0 -0
  38. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/persona/default.yaml +0 -0
  39. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/run-grounded.yaml +0 -0
  40. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/run.yaml +0 -0
  41. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/configs/safety_keywords.yaml +0 -0
  42. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/datasets/demo_conversations.jsonl +0 -0
  43. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/data/datasets/grounded_demo.jsonl +0 -0
  44. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/__init__.py +0 -0
  45. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/custom.py +0 -0
  46. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/evidence.py +0 -0
  47. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/faithfulness.py +0 -0
  48. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/grounding.py +0 -0
  49. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/evaluators/metrics.py +0 -0
  50. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/examples/__init__.py +0 -0
  51. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/examples/resource_task.py +0 -0
  52. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/__init__.py +0 -0
  53. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/archive.py +0 -0
  54. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/artifacts.py +0 -0
  55. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/comparison.py +0 -0
  56. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/evaluation.py +0 -0
  57. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/gates.py +0 -0
  58. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/leases.py +0 -0
  59. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/legacy.py +0 -0
  60. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/recovery.py +0 -0
  61. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/review.py +0 -0
  62. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/execution/suite.py +0 -0
  63. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/__init__.py +0 -0
  64. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/authenticity_judge.py +0 -0
  65. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/judges/prompts.py +0 -0
  66. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/__init__.py +0 -0
  67. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/anthropic.py +0 -0
  68. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/base.py +0 -0
  69. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/callable.py +0 -0
  70. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/classifiers.py +0 -0
  71. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/durable_judge.py +0 -0
  72. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/embeddings.py +0 -0
  73. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/judges.py +0 -0
  74. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/local.py +0 -0
  75. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/providers/openai.py +0 -0
  76. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/__init__.py +0 -0
  77. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/html.py +0 -0
  78. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/reporting/json_out.py +0 -0
  79. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/run_config.py +0 -0
  80. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/runner.py +0 -0
  81. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/__init__.py +0 -0
  82. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/evaluation.py +0 -0
  83. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/execution.py +0 -0
  84. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/gates.py +0 -0
  85. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/metrics.py +0 -0
  86. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/review.py +0 -0
  87. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/scoring.py +0 -0
  88. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/schemas/suite.py +0 -0
  89. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/__init__.py +0 -0
  90. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/authenticity.py +0 -0
  91. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/faithfulness.py +0 -0
  92. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/grounding.py +0 -0
  93. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/safety.py +0 -0
  94. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scorers/stability.py +0 -0
  95. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/__init__.py +0 -0
  96. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/bootstrap_dataset.py +0 -0
  97. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/calibrate_persona.py +0 -0
  98. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/run_openai_demo.py +0 -0
  99. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/scripts/sanitize_dataset.py +0 -0
  100. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/sdk.py +0 -0
  101. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/__init__.py +0 -0
  102. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/evaluations.py +0 -0
  103. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/reviews.py +0 -0
  104. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/storage/runs.py +0 -0
  105. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/__init__.py +0 -0
  106. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/io.py +0 -0
  107. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/optional.py +0 -0
  108. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/tokens.py +0 -0
  109. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter/utils/yaml.py +0 -0
  110. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/dependency_links.txt +0 -0
  111. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/entry_points.txt +0 -0
  112. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/requires.txt +0 -0
  113. {alignmenter-0.3.0 → alignmenter-0.3.1}/src/alignmenter.egg-info/top_level.txt +0 -0
  114. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/__init__.py +0 -0
  115. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/conftest.py +0 -0
  116. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_evaluation_judge.py +0 -0
  117. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_evaluation_worker.py +0 -0
  118. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_recovery_target.py +0 -0
  119. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_recovery_worker.py +0 -0
  120. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/durable_run_worker.py +0 -0
  121. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/data/mini_cli_dataset.jsonl +0 -0
  122. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_authenticity_judge.py +0 -0
  123. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_builtin_evaluations.py +0 -0
  124. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_calibrate_persona.py +0 -0
  125. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_capture_recovery.py +0 -0
  126. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_errors.py +0 -0
  127. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_grounded.py +0 -0
  128. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_helpers.py +0 -0
  129. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_import.py +0 -0
  130. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_init.py +0 -0
  131. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_cli_run_config.py +0 -0
  132. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_config.py +0 -0
  133. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_durable_evaluations.py +0 -0
  134. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_durable_execution.py +0 -0
  135. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_faithfulness.py +0 -0
  136. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_grounding.py +0 -0
  137. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_html_report.py +0 -0
  138. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_judge_providers.py +0 -0
  139. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_offline_safety.py +0 -0
  140. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_persona_gpt.py +0 -0
  141. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_provider_local.py +0 -0
  142. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_provider_openai.py +0 -0
  143. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_providers.py +0 -0
  144. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_release_workflow.py +0 -0
  145. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_review_workflow.py +0 -0
  146. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_config_grounded.py +0 -0
  147. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_config_loader.py +0 -0
  148. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_run_openai_demo.py +0 -0
  149. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_runner.py +0 -0
  150. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_sampling.py +0 -0
  151. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_scorers.py +0 -0
  152. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_smoke.py +0 -0
  153. {alignmenter-0.3.0 → alignmenter-0.3.1}/tests/test_suite_archive.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alignmenter
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Summary: Durable application alignment evaluations, evidence review, saved comparisons, and CI gates.
5
5
  Author: Alignmenter
6
6
  License-Expression: Apache-2.0
@@ -1,3 +1,3 @@
1
1
  """Single source for distribution and runtime version metadata."""
2
2
 
3
- __version__ = "0.3.0"
3
+ __version__ = "0.3.1"
@@ -22,6 +22,18 @@ from alignmenter.schemas.gates import GatePolicy
22
22
  EXIT_CODES = {"pass": 0, "fail": 2, "inconclusive": 3}
23
23
 
24
24
 
25
+ def _exit_code(decision: str, *, allow_inconclusive: bool = False) -> int:
26
+ """Map a decision to a process exit code.
27
+
28
+ With ``allow_inconclusive`` an inconclusive decision (e.g. an unreviewed
29
+ ``draft`` spec that met every applicable criterion) exits 0 so an automated
30
+ gate is not blocked; a genuine ``fail`` still exits non-zero.
31
+ """
32
+ if allow_inconclusive and decision == "inconclusive":
33
+ return 0
34
+ return EXIT_CODES[decision]
35
+
36
+
25
37
  def register_release_commands(app):
26
38
  @app.command("archive-export")
27
39
  def archive_export(
@@ -67,6 +79,9 @@ def register_release_commands(app):
67
79
  suite: Path = typer.Argument(..., exists=True, dir_okay=False),
68
80
  out: Path = typer.Option(Path("reports"), "--out"),
69
81
  resume: Path | None = typer.Option(None, "--resume", exists=True, file_okay=False),
82
+ allow_inconclusive: bool = typer.Option(
83
+ False, "--allow-inconclusive",
84
+ help="Exit 0 on an inconclusive decision (a fail still exits non-zero)."),
70
85
  ):
71
86
  """Capture, evaluate, compare, and write CI artifacts under a frozen suite config."""
72
87
  try:
@@ -74,7 +89,9 @@ def register_release_commands(app):
74
89
  except Exception as exc:
75
90
  raise typer.BadParameter(str(exc)) from exc
76
91
  typer.echo(json.dumps(result, indent=2))
77
- raise typer.Exit(EXIT_CODES[result["decision"]])
92
+ if allow_inconclusive and result["decision"] == "inconclusive":
93
+ typer.echo("Inconclusive tolerated (--allow-inconclusive): exiting 0.")
94
+ raise typer.Exit(_exit_code(result["decision"], allow_inconclusive=allow_inconclusive))
78
95
 
79
96
  @app.command("review-export")
80
97
  def review_export(
@@ -137,6 +154,9 @@ def register_release_commands(app):
137
154
  baseline: Path | None = typer.Option(None, "--baseline", exists=True, file_okay=False),
138
155
  baseline_id: UUID | None = typer.Option(None, "--baseline-id"),
139
156
  force: bool = typer.Option(False, "--force"),
157
+ allow_inconclusive: bool = typer.Option(
158
+ False, "--allow-inconclusive",
159
+ help="Exit 0 on an inconclusive decision (a fail still exits non-zero)."),
140
160
  ):
141
161
  """Check saved results and export CI artifacts without invoking any provider."""
142
162
  try:
@@ -147,7 +167,7 @@ def register_release_commands(app):
147
167
  raise typer.BadParameter(str(exc)) from exc
148
168
  decision = report["gate_report"]["decision"]
149
169
  typer.echo(f"Decision: {decision}\nArtifacts: {out.resolve()}")
150
- raise typer.Exit(EXIT_CODES[decision])
170
+ raise typer.Exit(_exit_code(decision, allow_inconclusive=allow_inconclusive))
151
171
 
152
172
  @app.command("compare")
153
173
  def compare(
@@ -12,6 +12,7 @@ from xml.etree import ElementTree as ET
12
12
 
13
13
  from alignmenter.execution.evaluation import evaluation_summary
14
14
  from alignmenter.execution.gates import gate_report
15
+ from alignmenter.reporting.github_comment import render_github_comment
15
16
 
16
17
 
17
18
  def _escape(value):
@@ -167,5 +168,6 @@ def export_evaluation(run_dir, out_dir, *, evaluation_id=None, policy=None, comp
167
168
  report["comparison"] = comparison
168
169
  report["review"] = qualification_report(run_dir, report["evaluation_id"])
169
170
  write_artifacts(out_dir, {"evaluation.json": _pretty(report) + "\n", "index.html": render_html(report),
170
- "junit.xml": render_junit(report), "summary.md": render_markdown(report)}, force=force)
171
+ "junit.xml": render_junit(report), "summary.md": render_markdown(report),
172
+ "comment.md": render_github_comment(report)}, force=force)
171
173
  return report
@@ -0,0 +1,140 @@
1
+ """Sticky GitHub pull-request comment over one saved decision; never executes a scorer.
2
+
3
+ Renders the same ``report`` structure the other durable reporters consume
4
+ (:mod:`alignmenter.reporting.durable`) into Markdown tuned for a PR comment: a
5
+ verdict badge, blocking issues first, a gate table, metrics (with baseline
6
+ deltas when a comparison is present), and collapsed breakdowns. The leading
7
+ marker lets a CI step find and update one sticky comment instead of appending.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ MARKER = "<!-- alignmenter:report -->"
13
+
14
+ _DECISION_BADGE = {"pass": "✅ Pass", "fail": "❌ Fail", "inconclusive": "⚠️ Inconclusive"}
15
+ _CHECK_ICON = {"pass": "✅", "fail": "❌", "inconclusive": "➖"}
16
+ _OPERATOR = {"at_least": "≥", "at_most": "≤"}
17
+
18
+
19
+ def _cell(value):
20
+ """Escape a value for a single Markdown table cell."""
21
+ if value is None:
22
+ return "—"
23
+ return (str(value).replace("\\", "\\\\").replace("|", "\\|")
24
+ .replace("\n", " ").replace("<", "&lt;").replace(">", "&gt;"))
25
+
26
+
27
+ def _num(value, digits=3):
28
+ if value is None:
29
+ return "—"
30
+ if isinstance(value, bool):
31
+ return str(value)
32
+ if isinstance(value, float):
33
+ return f"{value:.{digits}f}".rstrip("0").rstrip(".") or "0"
34
+ return str(value)
35
+
36
+
37
+ def _signed(value, digits=3):
38
+ if isinstance(value, (int, float)) and not isinstance(value, bool):
39
+ return ("+" if value > 0 else "") + _num(value, digits)
40
+ return _num(value, digits)
41
+
42
+
43
+ def _icon(decision):
44
+ return _CHECK_ICON.get(decision, _cell(decision))
45
+
46
+
47
+ def _gate_detail(check):
48
+ metric = check.get("metric")
49
+ if metric is None:
50
+ return _cell(check.get("reason", ""))
51
+ operator = _OPERATOR.get(check.get("operator"), check.get("operator") or "")
52
+ return f"`{_cell(metric)}` {operator} {_num(check.get('threshold'))} · got **{_num(check.get('value'))}**"
53
+
54
+
55
+ def _blocking_lines(report):
56
+ lines = []
57
+ for check in report["gate_report"].get("checks", []):
58
+ # Skip the structural "evaluation"/"comparison" umbrella checks (no metric);
59
+ # the ❌ badge and the violated cases below already convey those.
60
+ if check["decision"] == "fail" and check.get("metric") is not None:
61
+ lines.append(f"- **{_cell(check['id'])}** — {_gate_detail(check)}")
62
+ inputs = {item["key"]: item for item in report.get("inputs", [])}
63
+ violated = [(inputs.get(r["input_key"]), r) for r in report.get("results", [])
64
+ if r.get("status") == "violated"]
65
+ for item, result in violated[:5]:
66
+ item = item or {}
67
+ label = _cell(item.get("case_id") or item.get("session_id") or "case")
68
+ lines.append(f"- case **{label}** / {_cell(item.get('criterion_id'))}: "
69
+ f"{_cell(result.get('reason') or 'violated')}")
70
+ if len(violated) > 5:
71
+ lines.append(f"- …and {len(violated) - 5} more violated case(s)")
72
+ return lines
73
+
74
+
75
+ def _gate_table(report):
76
+ checks = report["gate_report"].get("checks", [])
77
+ if not checks:
78
+ return []
79
+ rows = ["| Gate | Result | Detail |", "| --- | :---: | --- |"]
80
+ rows += [f"| {_cell(c['id'])} | {_icon(c['decision'])} | {_gate_detail(c)} |" for c in checks]
81
+ return ["#### Gates", *rows, ""]
82
+
83
+
84
+ def _metrics_table(report):
85
+ comparison = report.get("comparison")
86
+ if comparison and comparison.get("metrics"):
87
+ rows = ["| Metric | Baseline | Candidate | Δ | Note |",
88
+ "| --- | ---: | ---: | ---: | :---: |"]
89
+ for name, metric in comparison["metrics"].items():
90
+ note = "⚠️ unavailable" if metric.get("unavailable") else ""
91
+ rows.append(f"| `{_cell(name)}` | {_num(metric.get('baseline'))} | "
92
+ f"{_num(metric.get('candidate'))} | {_signed(metric.get('delta'))} | {note} |")
93
+ return ["#### Metrics (vs baseline)", *rows, ""]
94
+ metrics = report.get("metrics") or {}
95
+ if not metrics:
96
+ return []
97
+ rows = ["| Metric | Value | Denominator |", "| --- | ---: | ---: |"]
98
+ rows += [f"| `{_cell(name)}` | {_num(m.get('value'))} | {_num(m.get('denominator'))} |"
99
+ for name, m in metrics.items()]
100
+ return ["#### Metrics", *rows, ""]
101
+
102
+
103
+ def _details(summary, body):
104
+ return [f"<details><summary>{_cell(summary)}</summary>", "", body, "", "</details>", ""]
105
+
106
+
107
+ def _breakdown(report):
108
+ lines = []
109
+ criteria = report.get("criteria") or {}
110
+ if criteria:
111
+ rows = ["| Criterion | met | violated | n/a | decision |",
112
+ "| --- | ---: | ---: | ---: | :---: |"]
113
+ for cid, summary in criteria.items():
114
+ counts = summary.get("counts", {})
115
+ rows.append(f"| `{_cell(cid)}` | {counts.get('met', 0)} | {counts.get('violated', 0)} | "
116
+ f"{counts.get('not_applicable', 0)} | {_icon(summary.get('decision'))} |")
117
+ lines += _details("Per-criterion breakdown", "\n".join(rows))
118
+ counts = report.get("counts") or {}
119
+ if counts:
120
+ lines += _details("Outcome counts", ", ".join(f"{k}: {v}" for k, v in counts.items()))
121
+ return lines
122
+
123
+
124
+ def render_github_comment(report, *, title="Alignmenter"):
125
+ """Return a sticky PR-comment Markdown body for one saved evaluation ``report``."""
126
+ decision = report["gate_report"]["decision"]
127
+ spec = report["spec"]
128
+ header = (f"`{_cell(spec['id'])} @ {_cell(spec['revision'])}` · "
129
+ f"qualification `{_cell(spec['qualification'])}` · "
130
+ f"{report['judged']}/{report['applicable']} applicable criteria evaluated")
131
+ if report.get("unavailable"):
132
+ header += f" · {report['unavailable']} unavailable"
133
+ lines = [MARKER, f"### {_DECISION_BADGE.get(decision, _cell(decision))} — {_cell(title)}", header, ""]
134
+ blocking = _blocking_lines(report)
135
+ if blocking:
136
+ lines += ["#### ❌ Blocking", *blocking, ""]
137
+ lines += _gate_table(report)
138
+ lines += _metrics_table(report)
139
+ lines += _breakdown(report)
140
+ return "\n".join(lines).rstrip() + "\n"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alignmenter
3
- Version: 0.3.0
3
+ Version: 0.3.1
4
4
  Summary: Durable application alignment evaluations, evidence review, saved comparisons, and CI gates.
5
5
  Author: Alignmenter
6
6
  License-Expression: Apache-2.0
@@ -77,6 +77,7 @@ src/alignmenter/providers/local.py
77
77
  src/alignmenter/providers/openai.py
78
78
  src/alignmenter/reporting/__init__.py
79
79
  src/alignmenter/reporting/durable.py
80
+ src/alignmenter/reporting/github_comment.py
80
81
  src/alignmenter/reporting/html.py
81
82
  src/alignmenter/reporting/json_out.py
82
83
  src/alignmenter/schemas/__init__.py
@@ -123,6 +124,7 @@ tests/test_config.py
123
124
  tests/test_durable_evaluations.py
124
125
  tests/test_durable_execution.py
125
126
  tests/test_faithfulness.py
127
+ tests/test_github_comment.py
126
128
  tests/test_grounding.py
127
129
  tests/test_html_report.py
128
130
  tests/test_judge_providers.py
@@ -0,0 +1,116 @@
1
+ """GitHub PR-comment reporter and the --allow-inconclusive CI exit flag."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typer.testing import CliRunner
6
+
7
+ from alignmenter.cli import app
8
+ from alignmenter.execution.evaluation import evaluate_saved
9
+ from alignmenter.release_cli import _exit_code
10
+ from alignmenter.reporting.durable import export_evaluation
11
+ from alignmenter.reporting.github_comment import MARKER, render_github_comment
12
+ from alignmenter.schemas.evaluation import JudgeBudget
13
+
14
+ from .test_durable_evaluations import Judge, captured, spec
15
+
16
+
17
+ def make_report(*, decision="pass", checks=None, metrics=None, comparison=None,
18
+ criteria=None, counts=None, inputs=None, results=None,
19
+ judged=2, applicable=2, unavailable=0, qualification="reviewed"):
20
+ return {
21
+ "spec": {"id": "avercare", "revision": "v1", "qualification": qualification},
22
+ "judged": judged, "applicable": applicable, "unavailable": unavailable,
23
+ "metrics": metrics or {}, "comparison": comparison,
24
+ "criteria": criteria or {}, "counts": counts or {},
25
+ "inputs": inputs or [], "results": results or [],
26
+ "gate_report": {"decision": decision, "checks": checks or [
27
+ {"id": "evaluation", "decision": decision, "reason": "All required outcomes."}]},
28
+ }
29
+
30
+
31
+ def test_pass_comment_has_marker_badge_and_header():
32
+ body = render_github_comment(make_report(decision="pass"))
33
+ assert body.splitlines()[0] == MARKER # first line so a sticky-comment step can match it
34
+ assert "✅ Pass" in body
35
+ assert "`avercare @ v1`" in body and "2/2 applicable criteria" in body
36
+
37
+
38
+ def test_failing_gate_and_violated_case_surface_in_blocking_section():
39
+ report = make_report(
40
+ decision="fail",
41
+ checks=[
42
+ {"id": "evaluation", "decision": "fail", "reason": "A criterion was violated."},
43
+ {"id": "dangerous_zero", "decision": "fail", "metric": "faithfulness.dangerous_answers",
44
+ "operator": "at_most", "threshold": 0, "value": 2},
45
+ ],
46
+ inputs=[{"key": "k1", "case_id": "hydration-01", "criterion_id": "faithfulness"}],
47
+ results=[{"input_key": "k1", "status": "violated", "reason": "Unsupported dosage claim."}],
48
+ )
49
+ body = render_github_comment(report)
50
+ blocking = body.split("#### ❌ Blocking", 1)[1]
51
+ assert "❌ Fail" in body
52
+ assert "**dangerous_zero**" in blocking and "≤ 0" in blocking and "got **2**" in blocking
53
+ assert "hydration-01" in blocking and "Unsupported dosage claim." in blocking
54
+
55
+
56
+ def test_inconclusive_badge_and_no_blocking_section():
57
+ body = render_github_comment(make_report(decision="inconclusive", qualification="draft"))
58
+ assert "⚠️ Inconclusive" in body
59
+ assert "#### ❌ Blocking" not in body
60
+
61
+
62
+ def test_comparison_renders_baseline_delta_columns():
63
+ report = make_report(
64
+ decision="pass",
65
+ comparison={"metrics": {
66
+ "evaluation.met_rate": {"baseline": 0.8, "candidate": 0.9, "delta": 0.1, "unavailable": False},
67
+ "grounding.citation_resolution": {"baseline": None, "candidate": None, "delta": None, "unavailable": True},
68
+ }},
69
+ )
70
+ body = render_github_comment(report)
71
+ assert "Metrics (vs baseline)" in body
72
+ assert "+0.1" in body # signed delta
73
+ assert "⚠️ unavailable" in body # unavailable metric flagged, not dropped
74
+
75
+
76
+ def test_table_cells_are_escaped_and_do_not_break_the_marker():
77
+ report = make_report(
78
+ decision="fail",
79
+ checks=[{"id": "evaluation", "decision": "fail", "reason": "pipe | and\nnewline <b>"}],
80
+ inputs=[{"key": "k1", "case_id": "a|b", "criterion_id": "c"}],
81
+ results=[{"input_key": "k1", "status": "violated", "reason": "x | y\nz"}],
82
+ )
83
+ body = render_github_comment(report)
84
+ assert body.splitlines()[0] == MARKER
85
+ # no raw pipe/newline leaks into a table cell that would break rendering
86
+ assert "a\\|b" in body and "x \\| y z" in body
87
+ assert "<b>" not in body
88
+
89
+
90
+ def test_exit_code_mapping_and_allow_inconclusive():
91
+ assert _exit_code("pass") == 0
92
+ assert _exit_code("fail") == 2
93
+ assert _exit_code("inconclusive") == 3
94
+ assert _exit_code("inconclusive", allow_inconclusive=True) == 0
95
+ assert _exit_code("fail", allow_inconclusive=True) == 2 # a real failure still blocks
96
+
97
+
98
+ def test_export_writes_comment_alongside_other_artifacts(tmp_path):
99
+ run_dir = captured(tmp_path)
100
+ evaluate_saved(run_dir, spec(), Judge(), budget=JudgeBudget(max_calls=10)) # reviewed + met => pass
101
+ out = tmp_path / "reports"
102
+ export_evaluation(run_dir, out)
103
+ for name in ("evaluation.json", "index.html", "junit.xml", "summary.md", "comment.md"):
104
+ assert (out / name).exists(), name
105
+ comment = (out / "comment.md").read_text()
106
+ assert comment.startswith(MARKER) and "✅ Pass" in comment
107
+
108
+
109
+ def test_check_cli_allow_inconclusive_flips_exit_but_not_failure(tmp_path):
110
+ run_dir = captured(tmp_path)
111
+ evaluate_saved(run_dir, spec(qualification="draft"), Judge(), budget=JudgeBudget(max_calls=10)) # met, but draft => inconclusive
112
+ default = CliRunner().invoke(app, ["check", str(run_dir), "--out", str(tmp_path / "r1")])
113
+ assert default.exit_code == 3, default.output
114
+ allowed = CliRunner().invoke(app, ["check", str(run_dir), "--out", str(tmp_path / "r2"), "--allow-inconclusive"])
115
+ assert allowed.exit_code == 0, allowed.output
116
+ assert (tmp_path / "r2" / "comment.md").exists()
File without changes
File without changes
File without changes
File without changes
File without changes