alignmenter 0.1.2__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {alignmenter-0.1.2/src/alignmenter.egg-info → alignmenter-0.2.0}/PKG-INFO +90 -47
  2. {alignmenter-0.1.2 → alignmenter-0.2.0}/README.md +63 -28
  3. alignmenter-0.2.0/pyproject.toml +110 -0
  4. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/analyze.py +5 -6
  5. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/bounds.py +19 -5
  6. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/diagnose.py +6 -8
  7. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/generate.py +3 -4
  8. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/label.py +4 -5
  9. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/optimize.py +23 -9
  10. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/validate.py +36 -21
  11. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/cli.py +119 -102
  12. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/config.py +13 -14
  13. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/authenticity_judge.py +21 -17
  14. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/__init__.py +2 -4
  15. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/anthropic.py +4 -4
  16. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/base.py +4 -4
  17. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/classifiers.py +3 -3
  18. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/embeddings.py +4 -5
  19. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/judges.py +60 -36
  20. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/local.py +6 -6
  21. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/openai.py +6 -6
  22. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/html.py +35 -7
  23. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/run_config.py +2 -2
  24. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/runner.py +29 -28
  25. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/authenticity.py +109 -12
  26. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/safety.py +15 -17
  27. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/stability.py +2 -2
  28. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/bootstrap_dataset.py +1 -2
  29. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/calibrate_persona.py +2 -3
  30. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/sanitize_dataset.py +4 -4
  31. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/io.py +2 -1
  32. alignmenter-0.2.0/src/alignmenter/utils/optional.py +21 -0
  33. {alignmenter-0.1.2 → alignmenter-0.2.0/src/alignmenter.egg-info}/PKG-INFO +90 -47
  34. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/SOURCES.txt +2 -0
  35. alignmenter-0.2.0/src/alignmenter.egg-info/requires.txt +31 -0
  36. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_helpers.py +1 -2
  37. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_run_config.py +1 -1
  38. alignmenter-0.2.0/tests/test_html_report.py +31 -0
  39. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_judge_providers.py +1 -2
  40. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_provider_openai.py +1 -1
  41. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_run_openai_demo.py +1 -0
  42. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_sampling.py +1 -1
  43. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_scorers.py +56 -3
  44. alignmenter-0.1.2/pyproject.toml +0 -67
  45. alignmenter-0.1.2/src/alignmenter.egg-info/requires.txt +0 -22
  46. {alignmenter-0.1.2 → alignmenter-0.2.0}/LICENSE +0 -0
  47. {alignmenter-0.1.2 → alignmenter-0.2.0}/setup.cfg +0 -0
  48. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/__init__.py +0 -0
  49. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/__init__.py +1 -1
  50. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/sampling.py +0 -0
  51. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/demo_config.yaml +0 -0
  52. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/judges/safety_prompt.txt +0 -0
  53. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/persona/default.yaml +0 -0
  54. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/run.yaml +0 -0
  55. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/safety_keywords.yaml +0 -0
  56. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/datasets/demo_conversations.jsonl +0 -0
  57. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/__init__.py +0 -0
  58. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/prompts.py +0 -0
  59. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/__init__.py +0 -0
  60. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/json_out.py +0 -0
  61. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/__init__.py +0 -0
  62. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/__init__.py +0 -0
  63. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/run_openai_demo.py +0 -0
  64. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/__init__.py +0 -0
  65. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/tokens.py +0 -0
  66. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/yaml.py +0 -0
  67. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/dependency_links.txt +0 -0
  68. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/entry_points.txt +0 -0
  69. {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/top_level.txt +0 -0
  70. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_authenticity_judge.py +0 -0
  71. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_calibrate_persona.py +0 -0
  72. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_errors.py +0 -0
  73. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_import.py +0 -0
  74. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_init.py +0 -0
  75. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_config.py +0 -0
  76. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_offline_safety.py +0 -0
  77. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_persona_gpt.py +0 -0
  78. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_provider_local.py +0 -0
  79. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_providers.py +0 -0
  80. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_run_config_loader.py +0 -0
  81. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_runner.py +0 -0
  82. {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_smoke.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alignmenter
3
- Version: 0.1.2
3
+ Version: 0.2.0
4
4
  Summary: Persona-aligned evaluation toolkit for auditing conversational AI authenticity, safety, and stability.
5
5
  Author: Alignmenter
6
6
  License: Apache License
@@ -209,8 +209,8 @@ Project-URL: Homepage, https://alignmenter.com
209
209
  Project-URL: Repository, https://github.com/justinGrosvenor/alignmenter
210
210
  Project-URL: Documentation, https://github.com/justinGrosvenor/alignmenter
211
211
  Project-URL: Bug Tracker, https://github.com/justinGrosvenor/alignmenter/issues
212
- Keywords: llm,evaluation,alignment,persona,safety
213
- Classifier: Development Status :: 3 - Alpha
212
+ Keywords: llm,evaluation,alignment,persona,safety,llm-judge
213
+ Classifier: Development Status :: 4 - Beta
214
214
  Classifier: Intended Audience :: Developers
215
215
  Classifier: Intended Audience :: Information Technology
216
216
  Classifier: License :: OSI Approved :: Apache Software License
@@ -219,35 +219,49 @@ Classifier: Programming Language :: Python
219
219
  Classifier: Programming Language :: Python :: 3
220
220
  Classifier: Programming Language :: Python :: 3.10
221
221
  Classifier: Programming Language :: Python :: 3.11
222
+ Classifier: Programming Language :: Python :: 3.12
223
+ Classifier: Programming Language :: Python :: 3.13
222
224
  Requires-Python: >=3.10
223
225
  Description-Content-Type: text/markdown
224
226
  License-File: LICENSE
225
- Requires-Dist: typer[all]
226
- Requires-Dist: pydantic
227
- Requires-Dist: pydantic-settings>=2.0
228
- Requires-Dist: openai>=1.12.0
229
- Requires-Dist: anthropic>=0.18.0
230
- Requires-Dist: scikit-learn>=1.3.0
231
- Requires-Dist: numpy>=1.24.0
232
- Requires-Dist: pyyaml>=6.0
233
- Requires-Dist: sentence-transformers>=2.2.2
234
- Requires-Dist: torch
235
- Requires-Dist: tiktoken>=0.7.0
236
- Requires-Dist: requests>=2.32.0
227
+ Requires-Dist: typer[all]<1,>=0.12
228
+ Requires-Dist: pydantic<3,>=2
229
+ Requires-Dist: pydantic-settings<3,>=2
230
+ Requires-Dist: openai<3,>=1.40
231
+ Requires-Dist: anthropic<1,>=0.40
232
+ Requires-Dist: pyyaml<7,>=6
233
+ Requires-Dist: tiktoken<1,>=0.7
234
+ Requires-Dist: requests<3,>=2.32
235
+ Provides-Extra: ml
236
+ Requires-Dist: torch<3,>=2; extra == "ml"
237
+ Requires-Dist: sentence-transformers<6,>=3; extra == "ml"
238
+ Requires-Dist: transformers<6,>=4.40; extra == "ml"
239
+ Provides-Extra: calibrate
240
+ Requires-Dist: scikit-learn<2,>=1.3; extra == "calibrate"
241
+ Requires-Dist: numpy<3,>=1.24; extra == "calibrate"
242
+ Provides-Extra: safety
243
+ Requires-Dist: alignmenter[ml]; extra == "safety"
244
+ Provides-Extra: all
245
+ Requires-Dist: alignmenter[calibrate,ml]; extra == "all"
237
246
  Provides-Extra: dev
238
- Requires-Dist: pytest; extra == "dev"
247
+ Requires-Dist: alignmenter[calibrate,ml]; extra == "dev"
248
+ Requires-Dist: pytest>=8; extra == "dev"
239
249
  Requires-Dist: pytest-cov; extra == "dev"
240
- Requires-Dist: ruff; extra == "dev"
250
+ Requires-Dist: ruff>=0.6; extra == "dev"
241
251
  Requires-Dist: build; extra == "dev"
242
252
  Requires-Dist: twine; extra == "dev"
243
- Provides-Extra: safety
244
- Requires-Dist: transformers>=4.30.0; extra == "safety"
245
253
  Dynamic: license-file
246
254
 
247
255
  <p align="center">
248
256
  <img src="https://alignmenter-branding.s3.us-west-2.amazonaws.com/alignmenter-banner.png" alt="Alignmenter" width="800">
249
257
  </p>
250
258
 
259
+ <p align="center">
260
+ <a href="https://pypi.org/project/alignmenter/"><img src="https://badge.fury.io/py/alignmenter.svg" alt="PyPI version"></a>
261
+ <a href="https://pepy.tech/project/alignmenter"><img src="https://pepy.tech/badge/alignmenter" alt="Downloads"></a>
262
+ <a href="https://github.com/justinGrosvenor/alignmenter/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="License"></a>
263
+ </p>
264
+
251
265
  <p align="center">
252
266
  <strong>Persona-aligned evaluation for conversational AI</strong>
253
267
  </p>
@@ -285,19 +299,37 @@ cd alignmenter
285
299
  python -m venv env
286
300
  source env/bin/activate # On Windows: env\Scripts\activate
287
301
 
288
- # Install with dev + safety extras
289
- pip install -e ./alignmenter[dev,safety]
302
+ # Install with all optional extras for development
303
+ pip install -e "./alignmenter[dev]"
290
304
  ```
291
305
 
292
306
  ### Install from PyPI
293
307
 
308
+ The core install is lightweight — no `torch`, no `scikit-learn`. It runs the
309
+ default scoring path (hashed embeddings + keyword safety, optionally blended
310
+ with an LLM judge) out of the box:
311
+
294
312
  ```bash
295
- pip install "alignmenter[safety]"
313
+ pip install alignmenter
296
314
  alignmenter init
315
+ alignmenter run --config configs/run.yaml
316
+ ```
317
+
318
+ Add optional extras only when you need them:
319
+
320
+ | Extra | Adds | Use it for |
321
+ |-------|------|------------|
322
+ | `alignmenter[ml]` | torch, sentence-transformers, transformers | Local embeddings (`--embedding sentence-transformer:...`) and the offline safety classifier |
323
+ | `alignmenter[calibrate]` | scikit-learn, numpy | The persona calibration pipeline (`calibrate*` commands) |
324
+ | `alignmenter[all]` | both of the above | Everything |
325
+
326
+ ```bash
327
+ # Example: better local embeddings + offline safety classifier
328
+ pip install "alignmenter[ml]"
297
329
  alignmenter run --config configs/run.yaml --embedding sentence-transformer:all-MiniLM-L6-v2
298
330
  ```
299
331
 
300
- > **Note**: The `safety` extra includes `transformers` for the offline safety classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
332
+ > **Offline safety classifier**: `[ml]` includes `transformers` for the offline classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
301
333
 
302
334
  ### Run Your First Evaluation
303
335
 
@@ -545,32 +577,43 @@ traits:
545
577
 
546
578
  ## API Usage
547
579
 
580
+ `Runner` coordinates transcript preparation, scoring, and report generation.
581
+ It takes a `RunConfig` plus a list of scorers, and `execute()` returns the
582
+ path to the timestamped report directory (JSON + HTML are written for you).
583
+
548
584
  ```python
549
- from alignmenter.runner import Runner
550
- from alignmenter.config import RunConfig
551
-
552
- # Load configuration
553
- config = RunConfig.from_yaml("configs/run/my_eval.yaml")
554
-
555
- # Execute evaluation
556
- runner = Runner(config)
557
- results = runner.execute()
558
-
559
- # Access scores
560
- print(f"Authenticity: {results['scores']['authenticity']['mean']:.3f}")
561
- print(f"Safety: {results['scores']['safety']['fused_judge']:.3f}")
562
- print(f"Stability: {results['scores']['stability']['session_variance']:.3f}")
563
-
564
- # Generate reports
565
- from alignmenter.reporting import HTMLReporter, JSONReporter
566
-
567
- html_reporter = HTMLReporter()
568
- html_reporter.write(
569
- run_dir=results["run_dir"],
570
- summary=results["summary"],
571
- scores=results["scores"],
572
- sessions=results["sessions"],
585
+ import json
586
+ from pathlib import Path
587
+
588
+ from alignmenter.runner import RunConfig, Runner
589
+ from alignmenter.scorers.authenticity import AuthenticityScorer
590
+ from alignmenter.scorers.safety import SafetyScorer
591
+ from alignmenter.scorers.stability import StabilityScorer
592
+
593
+ config = RunConfig(
594
+ model="openai:gpt-4o-mini",
595
+ dataset_path=Path("datasets/demo_conversations.jsonl"),
596
+ persona_path=Path("configs/persona/default.yaml"),
573
597
  )
598
+
599
+ # Pass a judge to AuthenticityScorer/SafetyScorer to blend LLM judgment in;
600
+ # omit it (as here) for a fully offline, deterministic run.
601
+ scorers = [
602
+ AuthenticityScorer(persona_path=config.persona_path, embedding="hashed"),
603
+ SafetyScorer(keyword_path=Path("configs/safety_keywords.yaml")),
604
+ StabilityScorer(embedding="hashed"),
605
+ ]
606
+
607
+ # generate_transcripts=False reuses recorded transcripts (no provider calls).
608
+ runner = Runner(config, scorers, generate_transcripts=False)
609
+ run_dir = runner.execute() # -> Path to reports/<timestamp>_<run_id>/
610
+
611
+ results = json.loads((run_dir / "results.json").read_text())
612
+ primary = results["scores"]["primary"]
613
+ auth = primary["authenticity"]
614
+ print(f"Authenticity: {auth['mean']:.3f} (basis: {auth['basis']})")
615
+ print(f"Safety: {primary['safety']['score']:.3f}")
616
+ print(f"Stability: {primary['stability']['stability']:.3f}")
574
617
  ```
575
618
 
576
619
  ## CI/CD Integration
@@ -2,6 +2,12 @@
2
2
  <img src="https://alignmenter-branding.s3.us-west-2.amazonaws.com/alignmenter-banner.png" alt="Alignmenter" width="800">
3
3
  </p>
4
4
 
5
+ <p align="center">
6
+ <a href="https://pypi.org/project/alignmenter/"><img src="https://badge.fury.io/py/alignmenter.svg" alt="PyPI version"></a>
7
+ <a href="https://pepy.tech/project/alignmenter"><img src="https://pepy.tech/badge/alignmenter" alt="Downloads"></a>
8
+ <a href="https://github.com/justinGrosvenor/alignmenter/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="License"></a>
9
+ </p>
10
+
5
11
  <p align="center">
6
12
  <strong>Persona-aligned evaluation for conversational AI</strong>
7
13
  </p>
@@ -39,19 +45,37 @@ cd alignmenter
39
45
  python -m venv env
40
46
  source env/bin/activate # On Windows: env\Scripts\activate
41
47
 
42
- # Install with dev + safety extras
43
- pip install -e ./alignmenter[dev,safety]
48
+ # Install with all optional extras for development
49
+ pip install -e "./alignmenter[dev]"
44
50
  ```
45
51
 
46
52
  ### Install from PyPI
47
53
 
54
+ The core install is lightweight — no `torch`, no `scikit-learn`. It runs the
55
+ default scoring path (hashed embeddings + keyword safety, optionally blended
56
+ with an LLM judge) out of the box:
57
+
48
58
  ```bash
49
- pip install "alignmenter[safety]"
59
+ pip install alignmenter
50
60
  alignmenter init
61
+ alignmenter run --config configs/run.yaml
62
+ ```
63
+
64
+ Add optional extras only when you need them:
65
+
66
+ | Extra | Adds | Use it for |
67
+ |-------|------|------------|
68
+ | `alignmenter[ml]` | torch, sentence-transformers, transformers | Local embeddings (`--embedding sentence-transformer:...`) and the offline safety classifier |
69
+ | `alignmenter[calibrate]` | scikit-learn, numpy | The persona calibration pipeline (`calibrate*` commands) |
70
+ | `alignmenter[all]` | both of the above | Everything |
71
+
72
+ ```bash
73
+ # Example: better local embeddings + offline safety classifier
74
+ pip install "alignmenter[ml]"
51
75
  alignmenter run --config configs/run.yaml --embedding sentence-transformer:all-MiniLM-L6-v2
52
76
  ```
53
77
 
54
- > **Note**: The `safety` extra includes `transformers` for the offline safety classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
78
+ > **Offline safety classifier**: `[ml]` includes `transformers` for the offline classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
55
79
 
56
80
  ### Run Your First Evaluation
57
81
 
@@ -299,32 +323,43 @@ traits:
299
323
 
300
324
  ## API Usage
301
325
 
326
+ `Runner` coordinates transcript preparation, scoring, and report generation.
327
+ It takes a `RunConfig` plus a list of scorers, and `execute()` returns the
328
+ path to the timestamped report directory (JSON + HTML are written for you).
329
+
302
330
  ```python
303
- from alignmenter.runner import Runner
304
- from alignmenter.config import RunConfig
305
-
306
- # Load configuration
307
- config = RunConfig.from_yaml("configs/run/my_eval.yaml")
308
-
309
- # Execute evaluation
310
- runner = Runner(config)
311
- results = runner.execute()
312
-
313
- # Access scores
314
- print(f"Authenticity: {results['scores']['authenticity']['mean']:.3f}")
315
- print(f"Safety: {results['scores']['safety']['fused_judge']:.3f}")
316
- print(f"Stability: {results['scores']['stability']['session_variance']:.3f}")
317
-
318
- # Generate reports
319
- from alignmenter.reporting import HTMLReporter, JSONReporter
320
-
321
- html_reporter = HTMLReporter()
322
- html_reporter.write(
323
- run_dir=results["run_dir"],
324
- summary=results["summary"],
325
- scores=results["scores"],
326
- sessions=results["sessions"],
331
+ import json
332
+ from pathlib import Path
333
+
334
+ from alignmenter.runner import RunConfig, Runner
335
+ from alignmenter.scorers.authenticity import AuthenticityScorer
336
+ from alignmenter.scorers.safety import SafetyScorer
337
+ from alignmenter.scorers.stability import StabilityScorer
338
+
339
+ config = RunConfig(
340
+ model="openai:gpt-4o-mini",
341
+ dataset_path=Path("datasets/demo_conversations.jsonl"),
342
+ persona_path=Path("configs/persona/default.yaml"),
327
343
  )
344
+
345
+ # Pass a judge to AuthenticityScorer/SafetyScorer to blend LLM judgment in;
346
+ # omit it (as here) for a fully offline, deterministic run.
347
+ scorers = [
348
+ AuthenticityScorer(persona_path=config.persona_path, embedding="hashed"),
349
+ SafetyScorer(keyword_path=Path("configs/safety_keywords.yaml")),
350
+ StabilityScorer(embedding="hashed"),
351
+ ]
352
+
353
+ # generate_transcripts=False reuses recorded transcripts (no provider calls).
354
+ runner = Runner(config, scorers, generate_transcripts=False)
355
+ run_dir = runner.execute() # -> Path to reports/<timestamp>_<run_id>/
356
+
357
+ results = json.loads((run_dir / "results.json").read_text())
358
+ primary = results["scores"]["primary"]
359
+ auth = primary["authenticity"]
360
+ print(f"Authenticity: {auth['mean']:.3f} (basis: {auth['basis']})")
361
+ print(f"Safety: {primary['safety']['score']:.3f}")
362
+ print(f"Stability: {primary['stability']['stability']:.3f}")
328
363
  ```
329
364
 
330
365
  ## CI/CD Integration
@@ -0,0 +1,110 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "alignmenter"
7
+ version = "0.2.0"
8
+ description = "Persona-aligned evaluation toolkit for auditing conversational AI authenticity, safety, and stability."
9
+ authors = [{name = "Alignmenter"}]
10
+ readme = {file = "README.md", content-type = "text/markdown"}
11
+ requires-python = ">=3.10"
12
+ license = {file = "LICENSE"}
13
+ keywords = ["llm", "evaluation", "alignment", "persona", "safety", "llm-judge"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Developers",
17
+ "Intended Audience :: Information Technology",
18
+ "License :: OSI Approved :: Apache Software License",
19
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
20
+ "Programming Language :: Python",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3.10",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ ]
27
+ # Core install is lightweight: no torch, no scikit-learn. The default scoring
28
+ # path (hashed embeddings + LLM judge + keyword safety) runs on these alone.
29
+ # Heavy local-model dependencies live in the optional [ml] and [calibrate] extras.
30
+ dependencies = [
31
+ "typer[all]>=0.12,<1",
32
+ "pydantic>=2,<3",
33
+ "pydantic-settings>=2,<3",
34
+ "openai>=1.40,<3",
35
+ "anthropic>=0.40,<1",
36
+ "pyyaml>=6,<7",
37
+ "tiktoken>=0.7,<1",
38
+ "requests>=2.32,<3",
39
+ ]
40
+
41
+ [project.optional-dependencies]
42
+ # Local embeddings (sentence-transformers) + offline safety classifier (transformers).
43
+ # This is the only extra that pulls in torch.
44
+ ml = [
45
+ "torch>=2,<3",
46
+ "sentence-transformers>=3,<6",
47
+ "transformers>=4.40,<6",
48
+ ]
49
+ # Persona calibration / trait-model training and diagnostics.
50
+ calibrate = [
51
+ "scikit-learn>=1.3,<2",
52
+ "numpy>=1.24,<3",
53
+ ]
54
+ # Back-compat alias: the offline safety classifier ships in [ml].
55
+ safety = ["alignmenter[ml]"]
56
+ # Everything runtime-optional in one shot.
57
+ all = ["alignmenter[ml,calibrate]"]
58
+ dev = [
59
+ "alignmenter[ml,calibrate]",
60
+ "pytest>=8",
61
+ "pytest-cov",
62
+ "ruff>=0.6",
63
+ "build",
64
+ "twine",
65
+ ]
66
+
67
+ [project.urls]
68
+ Homepage = "https://alignmenter.com"
69
+ Repository = "https://github.com/justinGrosvenor/alignmenter"
70
+ Documentation = "https://github.com/justinGrosvenor/alignmenter"
71
+ "Bug Tracker" = "https://github.com/justinGrosvenor/alignmenter/issues"
72
+
73
+ [project.scripts]
74
+ alignmenter = "alignmenter.cli:app"
75
+
76
+ [tool.setuptools]
77
+ include-package-data = true
78
+
79
+ [tool.setuptools.package-dir]
80
+ "" = "src"
81
+
82
+ [tool.setuptools.packages.find]
83
+ where = ["src"]
84
+
85
+ [tool.setuptools.package-data]
86
+ alignmenter = [
87
+ "data/configs/**/*.yaml",
88
+ "data/configs/**/*.txt",
89
+ "data/datasets/*.jsonl",
90
+ ]
91
+
92
+ [tool.ruff]
93
+ line-length = 100
94
+ target-version = "py310"
95
+ src = ["src", "tests"]
96
+
97
+ [tool.ruff.lint]
98
+ select = ["E", "F", "I", "UP", "B"]
99
+ ignore = ["E501", "B008"]
100
+
101
+ [tool.ruff.lint.per-file-ignores]
102
+ # Typer Exit/BadParameter are control-flow re-raises; chaining a traceback adds noise.
103
+ "src/alignmenter/cli.py" = ["B904"]
104
+ "src/alignmenter/scripts/run_openai_demo.py" = ["B904"]
105
+
106
+ [tool.pytest.ini_options]
107
+ testpaths = ["tests"]
108
+ filterwarnings = [
109
+ "error::DeprecationWarning:alignmenter.*",
110
+ ]
@@ -5,11 +5,10 @@ from __future__ import annotations
5
5
  import json
6
6
  from collections import defaultdict
7
7
  from pathlib import Path
8
- from typing import Optional
9
8
 
10
- from alignmenter.scorers.authenticity import AuthenticityScorer
11
- from alignmenter.providers.judges import load_judge_provider
12
9
  from alignmenter.judges.authenticity_judge import AuthenticityJudge
10
+ from alignmenter.providers.judges import load_judge_provider
11
+ from alignmenter.scorers.authenticity import AuthenticityScorer
13
12
 
14
13
 
15
14
  def analyze_scenario_performance(
@@ -17,10 +16,10 @@ def analyze_scenario_performance(
17
16
  persona_path: Path,
18
17
  output_path: Path,
19
18
  *,
20
- embedding_provider: Optional[str] = None,
19
+ embedding_provider: str | None = None,
21
20
  judge_provider: str,
22
21
  samples_per_scenario: int = 3,
23
- judge_budget: Optional[int] = None,
22
+ judge_budget: int | None = None,
24
23
  ) -> dict:
25
24
  """
26
25
  Analyze performance across different scenario types.
@@ -39,7 +38,7 @@ def analyze_scenario_performance(
39
38
  """
40
39
  # Load dataset
41
40
  sessions = []
42
- with open(dataset_path, "r") as f:
41
+ with open(dataset_path) as f:
43
42
  for line in f:
44
43
  if line.strip():
45
44
  session = json.loads(line)
@@ -5,12 +5,25 @@ from __future__ import annotations
5
5
  import json
6
6
  import math
7
7
  from pathlib import Path
8
- from typing import Optional
9
8
 
10
- import numpy as np
9
+ try:
10
+ import numpy as np
11
+ except ModuleNotFoundError as _exc: # pragma: no cover - exercised without [calibrate]
12
+ _CALIBRATE_IMPORT_ERROR: ModuleNotFoundError | None = _exc
13
+ np = None # type: ignore[assignment]
14
+ else:
15
+ _CALIBRATE_IMPORT_ERROR = None
11
16
 
12
17
  from alignmenter.providers.embeddings import load_embedding_provider
13
18
  from alignmenter.utils import load_yaml
19
+ from alignmenter.utils.optional import missing_dependency
20
+
21
+
22
+ def _require_calibrate() -> None:
23
+ if _CALIBRATE_IMPORT_ERROR is not None:
24
+ raise missing_dependency(
25
+ "Persona calibration", "calibrate", "scikit-learn, numpy"
26
+ ) from _CALIBRATE_IMPORT_ERROR
14
27
 
15
28
 
16
29
  def estimate_bounds(
@@ -18,7 +31,7 @@ def estimate_bounds(
18
31
  persona_path: Path,
19
32
  output_path: Path,
20
33
  *,
21
- embedding_provider: Optional[str] = None,
34
+ embedding_provider: str | None = None,
22
35
  percentile_low: float = 5.0,
23
36
  percentile_high: float = 95.0,
24
37
  ) -> dict:
@@ -38,9 +51,10 @@ def estimate_bounds(
38
51
  Returns:
39
52
  Bounds report with statistics
40
53
  """
54
+ _require_calibrate()
41
55
  # Load labeled data
42
56
  labeled_data = []
43
- with open(labeled_path, "r") as f:
57
+ with open(labeled_path) as f:
44
58
  for line in f:
45
59
  if line.strip():
46
60
  labeled_data.append(json.loads(line))
@@ -133,7 +147,7 @@ def estimate_bounds(
133
147
  with open(output_path, "w") as f:
134
148
  json.dump(report, f, indent=2)
135
149
 
136
- print(f"\n✓ Bounds estimation complete")
150
+ print("\n✓ Bounds estimation complete")
137
151
  print(f" Style similarity min: {style_sim_min:.4f}")
138
152
  print(f" Style similarity max: {style_sim_max:.4f}")
139
153
  print(f" Mean: {report['style_sim_mean']:.4f}")
@@ -4,12 +4,10 @@ from __future__ import annotations
4
4
 
5
5
  import json
6
6
  from pathlib import Path
7
- from typing import Optional
8
7
 
9
-
10
- from alignmenter.scorers.authenticity import AuthenticityScorer
11
- from alignmenter.providers.judges import load_judge_provider
12
8
  from alignmenter.judges.authenticity_judge import AuthenticityJudge
9
+ from alignmenter.providers.judges import load_judge_provider
10
+ from alignmenter.scorers.authenticity import AuthenticityScorer
13
11
 
14
12
 
15
13
  def diagnose_calibration_errors(
@@ -17,9 +15,9 @@ def diagnose_calibration_errors(
17
15
  persona_path: Path,
18
16
  output_path: Path,
19
17
  *,
20
- embedding_provider: Optional[str] = None,
18
+ embedding_provider: str | None = None,
21
19
  judge_provider: str,
22
- judge_budget: Optional[int] = None,
20
+ judge_budget: int | None = None,
23
21
  ) -> dict:
24
22
  """
25
23
  Diagnose calibration errors by analyzing false positives and false negatives.
@@ -37,7 +35,7 @@ def diagnose_calibration_errors(
37
35
  """
38
36
  # Load labeled data
39
37
  labeled_data = []
40
- with open(labeled_path, "r") as f:
38
+ with open(labeled_path) as f:
41
39
  for line in f:
42
40
  if line.strip():
43
41
  item = json.loads(line)
@@ -69,7 +67,7 @@ def diagnose_calibration_errors(
69
67
  # Identify errors
70
68
  false_positives = []
71
69
  false_negatives = []
72
- for i, (score, example) in enumerate(zip(scores, labeled_data)):
70
+ for i, (score, example) in enumerate(zip(scores, labeled_data, strict=False)):
73
71
  label = example["label"]
74
72
  prediction = 1 if score >= 0.5 else 0
75
73
 
@@ -6,7 +6,6 @@ import json
6
6
  import random
7
7
  from collections import defaultdict
8
8
  from pathlib import Path
9
- from typing import Optional
10
9
 
11
10
  from alignmenter.utils import load_yaml
12
11
 
@@ -42,7 +41,7 @@ def generate_candidates(
42
41
 
43
42
  # Read dataset
44
43
  records = []
45
- with open(dataset_path, "r") as f:
44
+ with open(dataset_path) as f:
46
45
  for line in f:
47
46
  if line.strip():
48
47
  records.append(json.loads(line))
@@ -126,7 +125,7 @@ def _sample_diverse(turns: list[dict], num_samples: int) -> list[dict]:
126
125
  samples_per_scenario = max(1, num_samples // len(scenarios))
127
126
  candidates = []
128
127
 
129
- for scenario, scenario_turns in by_scenario.items():
128
+ for _scenario, scenario_turns in by_scenario.items():
130
129
  n = min(samples_per_scenario, len(scenario_turns))
131
130
  candidates.extend(random.sample(scenario_turns, n))
132
131
 
@@ -225,7 +224,7 @@ def main():
225
224
  print(f"✓ Generated {result['total_candidates']} candidates")
226
225
  print(f" Strategy: {result['strategy']}")
227
226
  print(f" Output: {result['output_path']}")
228
- print(f"\nScenario distribution:")
227
+ print("\nScenario distribution:")
229
228
  for scenario, count in sorted(result['scenario_distribution'].items()):
230
229
  print(f" {scenario}: {count}")
231
230
 
@@ -6,7 +6,6 @@ import json
6
6
  import sys
7
7
  from datetime import datetime, timezone
8
8
  from pathlib import Path
9
- from typing import Optional
10
9
 
11
10
  from alignmenter.utils import load_yaml
12
11
 
@@ -17,7 +16,7 @@ def label_data(
17
16
  output_path: Path,
18
17
  *,
19
18
  append: bool = False,
20
- labeler: Optional[str] = None,
19
+ labeler: str | None = None,
21
20
  ) -> dict:
22
21
  """
23
22
  Interactively label responses as on-brand (1) or off-brand (0).
@@ -39,7 +38,7 @@ def label_data(
39
38
 
40
39
  # Load candidates
41
40
  candidates = []
42
- with open(input_path, "r") as f:
41
+ with open(input_path) as f:
43
42
  for line in f:
44
43
  if line.strip():
45
44
  candidates.append(json.loads(line))
@@ -51,7 +50,7 @@ def label_data(
51
50
  # Load existing labeled data if appending
52
51
  existing_texts = set()
53
52
  if append and output_path.exists():
54
- with open(output_path, "r") as f:
53
+ with open(output_path) as f:
55
54
  for line in f:
56
55
  if line.strip():
57
56
  item = json.loads(line)
@@ -165,7 +164,7 @@ def _print_persona_context(persona: dict):
165
164
  print(f" ... and {len(avoided) - 10} more")
166
165
 
167
166
 
168
- def _prompt_label() -> tuple[Optional[int], Optional[str], str]:
167
+ def _prompt_label() -> tuple[int | None, str | None, str]:
169
168
  """
170
169
  Prompt user for label, confidence, and notes.
171
170