alignmenter 0.1.2__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {alignmenter-0.1.2/src/alignmenter.egg-info → alignmenter-0.2.0}/PKG-INFO +90 -47
- {alignmenter-0.1.2 → alignmenter-0.2.0}/README.md +63 -28
- alignmenter-0.2.0/pyproject.toml +110 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/analyze.py +5 -6
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/bounds.py +19 -5
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/diagnose.py +6 -8
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/generate.py +3 -4
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/label.py +4 -5
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/optimize.py +23 -9
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/validate.py +36 -21
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/cli.py +119 -102
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/config.py +13 -14
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/authenticity_judge.py +21 -17
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/__init__.py +2 -4
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/anthropic.py +4 -4
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/base.py +4 -4
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/classifiers.py +3 -3
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/embeddings.py +4 -5
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/judges.py +60 -36
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/local.py +6 -6
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/providers/openai.py +6 -6
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/html.py +35 -7
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/run_config.py +2 -2
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/runner.py +29 -28
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/authenticity.py +109 -12
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/safety.py +15 -17
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/stability.py +2 -2
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/bootstrap_dataset.py +1 -2
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/calibrate_persona.py +2 -3
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/sanitize_dataset.py +4 -4
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/io.py +2 -1
- alignmenter-0.2.0/src/alignmenter/utils/optional.py +21 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0/src/alignmenter.egg-info}/PKG-INFO +90 -47
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/SOURCES.txt +2 -0
- alignmenter-0.2.0/src/alignmenter.egg-info/requires.txt +31 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_helpers.py +1 -2
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_run_config.py +1 -1
- alignmenter-0.2.0/tests/test_html_report.py +31 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_judge_providers.py +1 -2
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_provider_openai.py +1 -1
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_run_openai_demo.py +1 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_sampling.py +1 -1
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_scorers.py +56 -3
- alignmenter-0.1.2/pyproject.toml +0 -67
- alignmenter-0.1.2/src/alignmenter.egg-info/requires.txt +0 -22
- {alignmenter-0.1.2 → alignmenter-0.2.0}/LICENSE +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/setup.cfg +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/__init__.py +1 -1
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/calibration/sampling.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/demo_config.yaml +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/judges/safety_prompt.txt +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/persona/default.yaml +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/run.yaml +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/configs/safety_keywords.yaml +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/data/datasets/demo_conversations.jsonl +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/judges/prompts.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/reporting/json_out.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scorers/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/scripts/run_openai_demo.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/__init__.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/tokens.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter/utils/yaml.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/dependency_links.txt +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/entry_points.txt +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/src/alignmenter.egg-info/top_level.txt +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_authenticity_judge.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_calibrate_persona.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_errors.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_import.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_cli_init.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_config.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_offline_safety.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_persona_gpt.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_provider_local.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_providers.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_run_config_loader.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_runner.py +0 -0
- {alignmenter-0.1.2 → alignmenter-0.2.0}/tests/test_smoke.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: alignmenter
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Persona-aligned evaluation toolkit for auditing conversational AI authenticity, safety, and stability.
|
|
5
5
|
Author: Alignmenter
|
|
6
6
|
License: Apache License
|
|
@@ -209,8 +209,8 @@ Project-URL: Homepage, https://alignmenter.com
|
|
|
209
209
|
Project-URL: Repository, https://github.com/justinGrosvenor/alignmenter
|
|
210
210
|
Project-URL: Documentation, https://github.com/justinGrosvenor/alignmenter
|
|
211
211
|
Project-URL: Bug Tracker, https://github.com/justinGrosvenor/alignmenter/issues
|
|
212
|
-
Keywords: llm,evaluation,alignment,persona,safety
|
|
213
|
-
Classifier: Development Status ::
|
|
212
|
+
Keywords: llm,evaluation,alignment,persona,safety,llm-judge
|
|
213
|
+
Classifier: Development Status :: 4 - Beta
|
|
214
214
|
Classifier: Intended Audience :: Developers
|
|
215
215
|
Classifier: Intended Audience :: Information Technology
|
|
216
216
|
Classifier: License :: OSI Approved :: Apache Software License
|
|
@@ -219,35 +219,49 @@ Classifier: Programming Language :: Python
|
|
|
219
219
|
Classifier: Programming Language :: Python :: 3
|
|
220
220
|
Classifier: Programming Language :: Python :: 3.10
|
|
221
221
|
Classifier: Programming Language :: Python :: 3.11
|
|
222
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
223
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
222
224
|
Requires-Python: >=3.10
|
|
223
225
|
Description-Content-Type: text/markdown
|
|
224
226
|
License-File: LICENSE
|
|
225
|
-
Requires-Dist: typer[all]
|
|
226
|
-
Requires-Dist: pydantic
|
|
227
|
-
Requires-Dist: pydantic-settings
|
|
228
|
-
Requires-Dist: openai
|
|
229
|
-
Requires-Dist: anthropic
|
|
230
|
-
Requires-Dist:
|
|
231
|
-
Requires-Dist:
|
|
232
|
-
Requires-Dist:
|
|
233
|
-
|
|
234
|
-
Requires-Dist: torch
|
|
235
|
-
Requires-Dist:
|
|
236
|
-
Requires-Dist:
|
|
227
|
+
Requires-Dist: typer[all]<1,>=0.12
|
|
228
|
+
Requires-Dist: pydantic<3,>=2
|
|
229
|
+
Requires-Dist: pydantic-settings<3,>=2
|
|
230
|
+
Requires-Dist: openai<3,>=1.40
|
|
231
|
+
Requires-Dist: anthropic<1,>=0.40
|
|
232
|
+
Requires-Dist: pyyaml<7,>=6
|
|
233
|
+
Requires-Dist: tiktoken<1,>=0.7
|
|
234
|
+
Requires-Dist: requests<3,>=2.32
|
|
235
|
+
Provides-Extra: ml
|
|
236
|
+
Requires-Dist: torch<3,>=2; extra == "ml"
|
|
237
|
+
Requires-Dist: sentence-transformers<6,>=3; extra == "ml"
|
|
238
|
+
Requires-Dist: transformers<6,>=4.40; extra == "ml"
|
|
239
|
+
Provides-Extra: calibrate
|
|
240
|
+
Requires-Dist: scikit-learn<2,>=1.3; extra == "calibrate"
|
|
241
|
+
Requires-Dist: numpy<3,>=1.24; extra == "calibrate"
|
|
242
|
+
Provides-Extra: safety
|
|
243
|
+
Requires-Dist: alignmenter[ml]; extra == "safety"
|
|
244
|
+
Provides-Extra: all
|
|
245
|
+
Requires-Dist: alignmenter[calibrate,ml]; extra == "all"
|
|
237
246
|
Provides-Extra: dev
|
|
238
|
-
Requires-Dist:
|
|
247
|
+
Requires-Dist: alignmenter[calibrate,ml]; extra == "dev"
|
|
248
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
239
249
|
Requires-Dist: pytest-cov; extra == "dev"
|
|
240
|
-
Requires-Dist: ruff; extra == "dev"
|
|
250
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
241
251
|
Requires-Dist: build; extra == "dev"
|
|
242
252
|
Requires-Dist: twine; extra == "dev"
|
|
243
|
-
Provides-Extra: safety
|
|
244
|
-
Requires-Dist: transformers>=4.30.0; extra == "safety"
|
|
245
253
|
Dynamic: license-file
|
|
246
254
|
|
|
247
255
|
<p align="center">
|
|
248
256
|
<img src="https://alignmenter-branding.s3.us-west-2.amazonaws.com/alignmenter-banner.png" alt="Alignmenter" width="800">
|
|
249
257
|
</p>
|
|
250
258
|
|
|
259
|
+
<p align="center">
|
|
260
|
+
<a href="https://pypi.org/project/alignmenter/"><img src="https://badge.fury.io/py/alignmenter.svg" alt="PyPI version"></a>
|
|
261
|
+
<a href="https://pepy.tech/project/alignmenter"><img src="https://pepy.tech/badge/alignmenter" alt="Downloads"></a>
|
|
262
|
+
<a href="https://github.com/justinGrosvenor/alignmenter/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="License"></a>
|
|
263
|
+
</p>
|
|
264
|
+
|
|
251
265
|
<p align="center">
|
|
252
266
|
<strong>Persona-aligned evaluation for conversational AI</strong>
|
|
253
267
|
</p>
|
|
@@ -285,19 +299,37 @@ cd alignmenter
|
|
|
285
299
|
python -m venv env
|
|
286
300
|
source env/bin/activate # On Windows: env\Scripts\activate
|
|
287
301
|
|
|
288
|
-
# Install with
|
|
289
|
-
pip install -e ./alignmenter[dev
|
|
302
|
+
# Install with all optional extras for development
|
|
303
|
+
pip install -e "./alignmenter[dev]"
|
|
290
304
|
```
|
|
291
305
|
|
|
292
306
|
### Install from PyPI
|
|
293
307
|
|
|
308
|
+
The core install is lightweight — no `torch`, no `scikit-learn`. It runs the
|
|
309
|
+
default scoring path (hashed embeddings + keyword safety, optionally blended
|
|
310
|
+
with an LLM judge) out of the box:
|
|
311
|
+
|
|
294
312
|
```bash
|
|
295
|
-
pip install
|
|
313
|
+
pip install alignmenter
|
|
296
314
|
alignmenter init
|
|
315
|
+
alignmenter run --config configs/run.yaml
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
Add optional extras only when you need them:
|
|
319
|
+
|
|
320
|
+
| Extra | Adds | Use it for |
|
|
321
|
+
|-------|------|------------|
|
|
322
|
+
| `alignmenter[ml]` | torch, sentence-transformers, transformers | Local embeddings (`--embedding sentence-transformer:...`) and the offline safety classifier |
|
|
323
|
+
| `alignmenter[calibrate]` | scikit-learn, numpy | The persona calibration pipeline (`calibrate*` commands) |
|
|
324
|
+
| `alignmenter[all]` | both of the above | Everything |
|
|
325
|
+
|
|
326
|
+
```bash
|
|
327
|
+
# Example: better local embeddings + offline safety classifier
|
|
328
|
+
pip install "alignmenter[ml]"
|
|
297
329
|
alignmenter run --config configs/run.yaml --embedding sentence-transformer:all-MiniLM-L6-v2
|
|
298
330
|
```
|
|
299
331
|
|
|
300
|
-
> **
|
|
332
|
+
> **Offline safety classifier**: `[ml]` includes `transformers` for the offline classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
|
|
301
333
|
|
|
302
334
|
### Run Your First Evaluation
|
|
303
335
|
|
|
@@ -545,32 +577,43 @@ traits:
|
|
|
545
577
|
|
|
546
578
|
## API Usage
|
|
547
579
|
|
|
580
|
+
`Runner` coordinates transcript preparation, scoring, and report generation.
|
|
581
|
+
It takes a `RunConfig` plus a list of scorers, and `execute()` returns the
|
|
582
|
+
path to the timestamped report directory (JSON + HTML are written for you).
|
|
583
|
+
|
|
548
584
|
```python
|
|
549
|
-
|
|
550
|
-
from
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
print(f"Safety: {results['scores']['safety']['fused_judge']:.3f}")
|
|
562
|
-
print(f"Stability: {results['scores']['stability']['session_variance']:.3f}")
|
|
563
|
-
|
|
564
|
-
# Generate reports
|
|
565
|
-
from alignmenter.reporting import HTMLReporter, JSONReporter
|
|
566
|
-
|
|
567
|
-
html_reporter = HTMLReporter()
|
|
568
|
-
html_reporter.write(
|
|
569
|
-
run_dir=results["run_dir"],
|
|
570
|
-
summary=results["summary"],
|
|
571
|
-
scores=results["scores"],
|
|
572
|
-
sessions=results["sessions"],
|
|
585
|
+
import json
|
|
586
|
+
from pathlib import Path
|
|
587
|
+
|
|
588
|
+
from alignmenter.runner import RunConfig, Runner
|
|
589
|
+
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
590
|
+
from alignmenter.scorers.safety import SafetyScorer
|
|
591
|
+
from alignmenter.scorers.stability import StabilityScorer
|
|
592
|
+
|
|
593
|
+
config = RunConfig(
|
|
594
|
+
model="openai:gpt-4o-mini",
|
|
595
|
+
dataset_path=Path("datasets/demo_conversations.jsonl"),
|
|
596
|
+
persona_path=Path("configs/persona/default.yaml"),
|
|
573
597
|
)
|
|
598
|
+
|
|
599
|
+
# Pass a judge to AuthenticityScorer/SafetyScorer to blend LLM judgment in;
|
|
600
|
+
# omit it (as here) for a fully offline, deterministic run.
|
|
601
|
+
scorers = [
|
|
602
|
+
AuthenticityScorer(persona_path=config.persona_path, embedding="hashed"),
|
|
603
|
+
SafetyScorer(keyword_path=Path("configs/safety_keywords.yaml")),
|
|
604
|
+
StabilityScorer(embedding="hashed"),
|
|
605
|
+
]
|
|
606
|
+
|
|
607
|
+
# generate_transcripts=False reuses recorded transcripts (no provider calls).
|
|
608
|
+
runner = Runner(config, scorers, generate_transcripts=False)
|
|
609
|
+
run_dir = runner.execute() # -> Path to reports/<timestamp>_<run_id>/
|
|
610
|
+
|
|
611
|
+
results = json.loads((run_dir / "results.json").read_text())
|
|
612
|
+
primary = results["scores"]["primary"]
|
|
613
|
+
auth = primary["authenticity"]
|
|
614
|
+
print(f"Authenticity: {auth['mean']:.3f} (basis: {auth['basis']})")
|
|
615
|
+
print(f"Safety: {primary['safety']['score']:.3f}")
|
|
616
|
+
print(f"Stability: {primary['stability']['stability']:.3f}")
|
|
574
617
|
```
|
|
575
618
|
|
|
576
619
|
## CI/CD Integration
|
|
@@ -2,6 +2,12 @@
|
|
|
2
2
|
<img src="https://alignmenter-branding.s3.us-west-2.amazonaws.com/alignmenter-banner.png" alt="Alignmenter" width="800">
|
|
3
3
|
</p>
|
|
4
4
|
|
|
5
|
+
<p align="center">
|
|
6
|
+
<a href="https://pypi.org/project/alignmenter/"><img src="https://badge.fury.io/py/alignmenter.svg" alt="PyPI version"></a>
|
|
7
|
+
<a href="https://pepy.tech/project/alignmenter"><img src="https://pepy.tech/badge/alignmenter" alt="Downloads"></a>
|
|
8
|
+
<a href="https://github.com/justinGrosvenor/alignmenter/blob/main/LICENSE"><img src="https://img.shields.io/badge/License-Apache_2.0-blue.svg" alt="License"></a>
|
|
9
|
+
</p>
|
|
10
|
+
|
|
5
11
|
<p align="center">
|
|
6
12
|
<strong>Persona-aligned evaluation for conversational AI</strong>
|
|
7
13
|
</p>
|
|
@@ -39,19 +45,37 @@ cd alignmenter
|
|
|
39
45
|
python -m venv env
|
|
40
46
|
source env/bin/activate # On Windows: env\Scripts\activate
|
|
41
47
|
|
|
42
|
-
# Install with
|
|
43
|
-
pip install -e ./alignmenter[dev
|
|
48
|
+
# Install with all optional extras for development
|
|
49
|
+
pip install -e "./alignmenter[dev]"
|
|
44
50
|
```
|
|
45
51
|
|
|
46
52
|
### Install from PyPI
|
|
47
53
|
|
|
54
|
+
The core install is lightweight — no `torch`, no `scikit-learn`. It runs the
|
|
55
|
+
default scoring path (hashed embeddings + keyword safety, optionally blended
|
|
56
|
+
with an LLM judge) out of the box:
|
|
57
|
+
|
|
48
58
|
```bash
|
|
49
|
-
pip install
|
|
59
|
+
pip install alignmenter
|
|
50
60
|
alignmenter init
|
|
61
|
+
alignmenter run --config configs/run.yaml
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Add optional extras only when you need them:
|
|
65
|
+
|
|
66
|
+
| Extra | Adds | Use it for |
|
|
67
|
+
|-------|------|------------|
|
|
68
|
+
| `alignmenter[ml]` | torch, sentence-transformers, transformers | Local embeddings (`--embedding sentence-transformer:...`) and the offline safety classifier |
|
|
69
|
+
| `alignmenter[calibrate]` | scikit-learn, numpy | The persona calibration pipeline (`calibrate*` commands) |
|
|
70
|
+
| `alignmenter[all]` | both of the above | Everything |
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
# Example: better local embeddings + offline safety classifier
|
|
74
|
+
pip install "alignmenter[ml]"
|
|
51
75
|
alignmenter run --config configs/run.yaml --embedding sentence-transformer:all-MiniLM-L6-v2
|
|
52
76
|
```
|
|
53
77
|
|
|
54
|
-
> **
|
|
78
|
+
> **Offline safety classifier**: `[ml]` includes `transformers` for the offline classifier (ProtectAI/distilled-safety-roberta). Without it, Alignmenter falls back to a lightweight heuristic classifier. See [docs/offline_safety.md](https://github.com/justinGrosvenor/alignmenter/blob/main/docs/offline_safety.md) for details.
|
|
55
79
|
|
|
56
80
|
### Run Your First Evaluation
|
|
57
81
|
|
|
@@ -299,32 +323,43 @@ traits:
|
|
|
299
323
|
|
|
300
324
|
## API Usage
|
|
301
325
|
|
|
326
|
+
`Runner` coordinates transcript preparation, scoring, and report generation.
|
|
327
|
+
It takes a `RunConfig` plus a list of scorers, and `execute()` returns the
|
|
328
|
+
path to the timestamped report directory (JSON + HTML are written for you).
|
|
329
|
+
|
|
302
330
|
```python
|
|
303
|
-
|
|
304
|
-
from
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
print(f"Safety: {results['scores']['safety']['fused_judge']:.3f}")
|
|
316
|
-
print(f"Stability: {results['scores']['stability']['session_variance']:.3f}")
|
|
317
|
-
|
|
318
|
-
# Generate reports
|
|
319
|
-
from alignmenter.reporting import HTMLReporter, JSONReporter
|
|
320
|
-
|
|
321
|
-
html_reporter = HTMLReporter()
|
|
322
|
-
html_reporter.write(
|
|
323
|
-
run_dir=results["run_dir"],
|
|
324
|
-
summary=results["summary"],
|
|
325
|
-
scores=results["scores"],
|
|
326
|
-
sessions=results["sessions"],
|
|
331
|
+
import json
|
|
332
|
+
from pathlib import Path
|
|
333
|
+
|
|
334
|
+
from alignmenter.runner import RunConfig, Runner
|
|
335
|
+
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
336
|
+
from alignmenter.scorers.safety import SafetyScorer
|
|
337
|
+
from alignmenter.scorers.stability import StabilityScorer
|
|
338
|
+
|
|
339
|
+
config = RunConfig(
|
|
340
|
+
model="openai:gpt-4o-mini",
|
|
341
|
+
dataset_path=Path("datasets/demo_conversations.jsonl"),
|
|
342
|
+
persona_path=Path("configs/persona/default.yaml"),
|
|
327
343
|
)
|
|
344
|
+
|
|
345
|
+
# Pass a judge to AuthenticityScorer/SafetyScorer to blend LLM judgment in;
|
|
346
|
+
# omit it (as here) for a fully offline, deterministic run.
|
|
347
|
+
scorers = [
|
|
348
|
+
AuthenticityScorer(persona_path=config.persona_path, embedding="hashed"),
|
|
349
|
+
SafetyScorer(keyword_path=Path("configs/safety_keywords.yaml")),
|
|
350
|
+
StabilityScorer(embedding="hashed"),
|
|
351
|
+
]
|
|
352
|
+
|
|
353
|
+
# generate_transcripts=False reuses recorded transcripts (no provider calls).
|
|
354
|
+
runner = Runner(config, scorers, generate_transcripts=False)
|
|
355
|
+
run_dir = runner.execute() # -> Path to reports/<timestamp>_<run_id>/
|
|
356
|
+
|
|
357
|
+
results = json.loads((run_dir / "results.json").read_text())
|
|
358
|
+
primary = results["scores"]["primary"]
|
|
359
|
+
auth = primary["authenticity"]
|
|
360
|
+
print(f"Authenticity: {auth['mean']:.3f} (basis: {auth['basis']})")
|
|
361
|
+
print(f"Safety: {primary['safety']['score']:.3f}")
|
|
362
|
+
print(f"Stability: {primary['stability']['stability']:.3f}")
|
|
328
363
|
```
|
|
329
364
|
|
|
330
365
|
## CI/CD Integration
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "alignmenter"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Persona-aligned evaluation toolkit for auditing conversational AI authenticity, safety, and stability."
|
|
9
|
+
authors = [{name = "Alignmenter"}]
|
|
10
|
+
readme = {file = "README.md", content-type = "text/markdown"}
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
license = {file = "LICENSE"}
|
|
13
|
+
keywords = ["llm", "evaluation", "alignment", "persona", "safety", "llm-judge"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"Intended Audience :: Information Technology",
|
|
18
|
+
"License :: OSI Approved :: Apache Software License",
|
|
19
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
20
|
+
"Programming Language :: Python",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.10",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
]
|
|
27
|
+
# Core install is lightweight: no torch, no scikit-learn. The default scoring
|
|
28
|
+
# path (hashed embeddings + LLM judge + keyword safety) runs on these alone.
|
|
29
|
+
# Heavy local-model dependencies live in the optional [ml] and [calibrate] extras.
|
|
30
|
+
dependencies = [
|
|
31
|
+
"typer[all]>=0.12,<1",
|
|
32
|
+
"pydantic>=2,<3",
|
|
33
|
+
"pydantic-settings>=2,<3",
|
|
34
|
+
"openai>=1.40,<3",
|
|
35
|
+
"anthropic>=0.40,<1",
|
|
36
|
+
"pyyaml>=6,<7",
|
|
37
|
+
"tiktoken>=0.7,<1",
|
|
38
|
+
"requests>=2.32,<3",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
# Local embeddings (sentence-transformers) + offline safety classifier (transformers).
|
|
43
|
+
# This is the only extra that pulls in torch.
|
|
44
|
+
ml = [
|
|
45
|
+
"torch>=2,<3",
|
|
46
|
+
"sentence-transformers>=3,<6",
|
|
47
|
+
"transformers>=4.40,<6",
|
|
48
|
+
]
|
|
49
|
+
# Persona calibration / trait-model training and diagnostics.
|
|
50
|
+
calibrate = [
|
|
51
|
+
"scikit-learn>=1.3,<2",
|
|
52
|
+
"numpy>=1.24,<3",
|
|
53
|
+
]
|
|
54
|
+
# Back-compat alias: the offline safety classifier ships in [ml].
|
|
55
|
+
safety = ["alignmenter[ml]"]
|
|
56
|
+
# Everything runtime-optional in one shot.
|
|
57
|
+
all = ["alignmenter[ml,calibrate]"]
|
|
58
|
+
dev = [
|
|
59
|
+
"alignmenter[ml,calibrate]",
|
|
60
|
+
"pytest>=8",
|
|
61
|
+
"pytest-cov",
|
|
62
|
+
"ruff>=0.6",
|
|
63
|
+
"build",
|
|
64
|
+
"twine",
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
[project.urls]
|
|
68
|
+
Homepage = "https://alignmenter.com"
|
|
69
|
+
Repository = "https://github.com/justinGrosvenor/alignmenter"
|
|
70
|
+
Documentation = "https://github.com/justinGrosvenor/alignmenter"
|
|
71
|
+
"Bug Tracker" = "https://github.com/justinGrosvenor/alignmenter/issues"
|
|
72
|
+
|
|
73
|
+
[project.scripts]
|
|
74
|
+
alignmenter = "alignmenter.cli:app"
|
|
75
|
+
|
|
76
|
+
[tool.setuptools]
|
|
77
|
+
include-package-data = true
|
|
78
|
+
|
|
79
|
+
[tool.setuptools.package-dir]
|
|
80
|
+
"" = "src"
|
|
81
|
+
|
|
82
|
+
[tool.setuptools.packages.find]
|
|
83
|
+
where = ["src"]
|
|
84
|
+
|
|
85
|
+
[tool.setuptools.package-data]
|
|
86
|
+
alignmenter = [
|
|
87
|
+
"data/configs/**/*.yaml",
|
|
88
|
+
"data/configs/**/*.txt",
|
|
89
|
+
"data/datasets/*.jsonl",
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
[tool.ruff]
|
|
93
|
+
line-length = 100
|
|
94
|
+
target-version = "py310"
|
|
95
|
+
src = ["src", "tests"]
|
|
96
|
+
|
|
97
|
+
[tool.ruff.lint]
|
|
98
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
99
|
+
ignore = ["E501", "B008"]
|
|
100
|
+
|
|
101
|
+
[tool.ruff.lint.per-file-ignores]
|
|
102
|
+
# Typer Exit/BadParameter are control-flow re-raises; chaining a traceback adds noise.
|
|
103
|
+
"src/alignmenter/cli.py" = ["B904"]
|
|
104
|
+
"src/alignmenter/scripts/run_openai_demo.py" = ["B904"]
|
|
105
|
+
|
|
106
|
+
[tool.pytest.ini_options]
|
|
107
|
+
testpaths = ["tests"]
|
|
108
|
+
filterwarnings = [
|
|
109
|
+
"error::DeprecationWarning:alignmenter.*",
|
|
110
|
+
]
|
|
@@ -5,11 +5,10 @@ from __future__ import annotations
|
|
|
5
5
|
import json
|
|
6
6
|
from collections import defaultdict
|
|
7
7
|
from pathlib import Path
|
|
8
|
-
from typing import Optional
|
|
9
8
|
|
|
10
|
-
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
11
|
-
from alignmenter.providers.judges import load_judge_provider
|
|
12
9
|
from alignmenter.judges.authenticity_judge import AuthenticityJudge
|
|
10
|
+
from alignmenter.providers.judges import load_judge_provider
|
|
11
|
+
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
13
12
|
|
|
14
13
|
|
|
15
14
|
def analyze_scenario_performance(
|
|
@@ -17,10 +16,10 @@ def analyze_scenario_performance(
|
|
|
17
16
|
persona_path: Path,
|
|
18
17
|
output_path: Path,
|
|
19
18
|
*,
|
|
20
|
-
embedding_provider:
|
|
19
|
+
embedding_provider: str | None = None,
|
|
21
20
|
judge_provider: str,
|
|
22
21
|
samples_per_scenario: int = 3,
|
|
23
|
-
judge_budget:
|
|
22
|
+
judge_budget: int | None = None,
|
|
24
23
|
) -> dict:
|
|
25
24
|
"""
|
|
26
25
|
Analyze performance across different scenario types.
|
|
@@ -39,7 +38,7 @@ def analyze_scenario_performance(
|
|
|
39
38
|
"""
|
|
40
39
|
# Load dataset
|
|
41
40
|
sessions = []
|
|
42
|
-
with open(dataset_path
|
|
41
|
+
with open(dataset_path) as f:
|
|
43
42
|
for line in f:
|
|
44
43
|
if line.strip():
|
|
45
44
|
session = json.loads(line)
|
|
@@ -5,12 +5,25 @@ from __future__ import annotations
|
|
|
5
5
|
import json
|
|
6
6
|
import math
|
|
7
7
|
from pathlib import Path
|
|
8
|
-
from typing import Optional
|
|
9
8
|
|
|
10
|
-
|
|
9
|
+
try:
|
|
10
|
+
import numpy as np
|
|
11
|
+
except ModuleNotFoundError as _exc: # pragma: no cover - exercised without [calibrate]
|
|
12
|
+
_CALIBRATE_IMPORT_ERROR: ModuleNotFoundError | None = _exc
|
|
13
|
+
np = None # type: ignore[assignment]
|
|
14
|
+
else:
|
|
15
|
+
_CALIBRATE_IMPORT_ERROR = None
|
|
11
16
|
|
|
12
17
|
from alignmenter.providers.embeddings import load_embedding_provider
|
|
13
18
|
from alignmenter.utils import load_yaml
|
|
19
|
+
from alignmenter.utils.optional import missing_dependency
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _require_calibrate() -> None:
|
|
23
|
+
if _CALIBRATE_IMPORT_ERROR is not None:
|
|
24
|
+
raise missing_dependency(
|
|
25
|
+
"Persona calibration", "calibrate", "scikit-learn, numpy"
|
|
26
|
+
) from _CALIBRATE_IMPORT_ERROR
|
|
14
27
|
|
|
15
28
|
|
|
16
29
|
def estimate_bounds(
|
|
@@ -18,7 +31,7 @@ def estimate_bounds(
|
|
|
18
31
|
persona_path: Path,
|
|
19
32
|
output_path: Path,
|
|
20
33
|
*,
|
|
21
|
-
embedding_provider:
|
|
34
|
+
embedding_provider: str | None = None,
|
|
22
35
|
percentile_low: float = 5.0,
|
|
23
36
|
percentile_high: float = 95.0,
|
|
24
37
|
) -> dict:
|
|
@@ -38,9 +51,10 @@ def estimate_bounds(
|
|
|
38
51
|
Returns:
|
|
39
52
|
Bounds report with statistics
|
|
40
53
|
"""
|
|
54
|
+
_require_calibrate()
|
|
41
55
|
# Load labeled data
|
|
42
56
|
labeled_data = []
|
|
43
|
-
with open(labeled_path
|
|
57
|
+
with open(labeled_path) as f:
|
|
44
58
|
for line in f:
|
|
45
59
|
if line.strip():
|
|
46
60
|
labeled_data.append(json.loads(line))
|
|
@@ -133,7 +147,7 @@ def estimate_bounds(
|
|
|
133
147
|
with open(output_path, "w") as f:
|
|
134
148
|
json.dump(report, f, indent=2)
|
|
135
149
|
|
|
136
|
-
print(
|
|
150
|
+
print("\n✓ Bounds estimation complete")
|
|
137
151
|
print(f" Style similarity min: {style_sim_min:.4f}")
|
|
138
152
|
print(f" Style similarity max: {style_sim_max:.4f}")
|
|
139
153
|
print(f" Mean: {report['style_sim_mean']:.4f}")
|
|
@@ -4,12 +4,10 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
6
|
from pathlib import Path
|
|
7
|
-
from typing import Optional
|
|
8
7
|
|
|
9
|
-
|
|
10
|
-
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
11
|
-
from alignmenter.providers.judges import load_judge_provider
|
|
12
8
|
from alignmenter.judges.authenticity_judge import AuthenticityJudge
|
|
9
|
+
from alignmenter.providers.judges import load_judge_provider
|
|
10
|
+
from alignmenter.scorers.authenticity import AuthenticityScorer
|
|
13
11
|
|
|
14
12
|
|
|
15
13
|
def diagnose_calibration_errors(
|
|
@@ -17,9 +15,9 @@ def diagnose_calibration_errors(
|
|
|
17
15
|
persona_path: Path,
|
|
18
16
|
output_path: Path,
|
|
19
17
|
*,
|
|
20
|
-
embedding_provider:
|
|
18
|
+
embedding_provider: str | None = None,
|
|
21
19
|
judge_provider: str,
|
|
22
|
-
judge_budget:
|
|
20
|
+
judge_budget: int | None = None,
|
|
23
21
|
) -> dict:
|
|
24
22
|
"""
|
|
25
23
|
Diagnose calibration errors by analyzing false positives and false negatives.
|
|
@@ -37,7 +35,7 @@ def diagnose_calibration_errors(
|
|
|
37
35
|
"""
|
|
38
36
|
# Load labeled data
|
|
39
37
|
labeled_data = []
|
|
40
|
-
with open(labeled_path
|
|
38
|
+
with open(labeled_path) as f:
|
|
41
39
|
for line in f:
|
|
42
40
|
if line.strip():
|
|
43
41
|
item = json.loads(line)
|
|
@@ -69,7 +67,7 @@ def diagnose_calibration_errors(
|
|
|
69
67
|
# Identify errors
|
|
70
68
|
false_positives = []
|
|
71
69
|
false_negatives = []
|
|
72
|
-
for i, (score, example) in enumerate(zip(scores, labeled_data)):
|
|
70
|
+
for i, (score, example) in enumerate(zip(scores, labeled_data, strict=False)):
|
|
73
71
|
label = example["label"]
|
|
74
72
|
prediction = 1 if score >= 0.5 else 0
|
|
75
73
|
|
|
@@ -6,7 +6,6 @@ import json
|
|
|
6
6
|
import random
|
|
7
7
|
from collections import defaultdict
|
|
8
8
|
from pathlib import Path
|
|
9
|
-
from typing import Optional
|
|
10
9
|
|
|
11
10
|
from alignmenter.utils import load_yaml
|
|
12
11
|
|
|
@@ -42,7 +41,7 @@ def generate_candidates(
|
|
|
42
41
|
|
|
43
42
|
# Read dataset
|
|
44
43
|
records = []
|
|
45
|
-
with open(dataset_path
|
|
44
|
+
with open(dataset_path) as f:
|
|
46
45
|
for line in f:
|
|
47
46
|
if line.strip():
|
|
48
47
|
records.append(json.loads(line))
|
|
@@ -126,7 +125,7 @@ def _sample_diverse(turns: list[dict], num_samples: int) -> list[dict]:
|
|
|
126
125
|
samples_per_scenario = max(1, num_samples // len(scenarios))
|
|
127
126
|
candidates = []
|
|
128
127
|
|
|
129
|
-
for
|
|
128
|
+
for _scenario, scenario_turns in by_scenario.items():
|
|
130
129
|
n = min(samples_per_scenario, len(scenario_turns))
|
|
131
130
|
candidates.extend(random.sample(scenario_turns, n))
|
|
132
131
|
|
|
@@ -225,7 +224,7 @@ def main():
|
|
|
225
224
|
print(f"✓ Generated {result['total_candidates']} candidates")
|
|
226
225
|
print(f" Strategy: {result['strategy']}")
|
|
227
226
|
print(f" Output: {result['output_path']}")
|
|
228
|
-
print(
|
|
227
|
+
print("\nScenario distribution:")
|
|
229
228
|
for scenario, count in sorted(result['scenario_distribution'].items()):
|
|
230
229
|
print(f" {scenario}: {count}")
|
|
231
230
|
|
|
@@ -6,7 +6,6 @@ import json
|
|
|
6
6
|
import sys
|
|
7
7
|
from datetime import datetime, timezone
|
|
8
8
|
from pathlib import Path
|
|
9
|
-
from typing import Optional
|
|
10
9
|
|
|
11
10
|
from alignmenter.utils import load_yaml
|
|
12
11
|
|
|
@@ -17,7 +16,7 @@ def label_data(
|
|
|
17
16
|
output_path: Path,
|
|
18
17
|
*,
|
|
19
18
|
append: bool = False,
|
|
20
|
-
labeler:
|
|
19
|
+
labeler: str | None = None,
|
|
21
20
|
) -> dict:
|
|
22
21
|
"""
|
|
23
22
|
Interactively label responses as on-brand (1) or off-brand (0).
|
|
@@ -39,7 +38,7 @@ def label_data(
|
|
|
39
38
|
|
|
40
39
|
# Load candidates
|
|
41
40
|
candidates = []
|
|
42
|
-
with open(input_path
|
|
41
|
+
with open(input_path) as f:
|
|
43
42
|
for line in f:
|
|
44
43
|
if line.strip():
|
|
45
44
|
candidates.append(json.loads(line))
|
|
@@ -51,7 +50,7 @@ def label_data(
|
|
|
51
50
|
# Load existing labeled data if appending
|
|
52
51
|
existing_texts = set()
|
|
53
52
|
if append and output_path.exists():
|
|
54
|
-
with open(output_path
|
|
53
|
+
with open(output_path) as f:
|
|
55
54
|
for line in f:
|
|
56
55
|
if line.strip():
|
|
57
56
|
item = json.loads(line)
|
|
@@ -165,7 +164,7 @@ def _print_persona_context(persona: dict):
|
|
|
165
164
|
print(f" ... and {len(avoided) - 10} more")
|
|
166
165
|
|
|
167
166
|
|
|
168
|
-
def _prompt_label() -> tuple[
|
|
167
|
+
def _prompt_label() -> tuple[int | None, str | None, str]:
|
|
169
168
|
"""
|
|
170
169
|
Prompt user for label, confidence, and notes.
|
|
171
170
|
|