promptgold 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- promptgold-0.1.0/.github/workflows/ci.yml +24 -0
- promptgold-0.1.0/.gitignore +9 -0
- promptgold-0.1.0/CONTRIBUTING.md +33 -0
- promptgold-0.1.0/LICENSE +21 -0
- promptgold-0.1.0/PKG-INFO +155 -0
- promptgold-0.1.0/README.md +123 -0
- promptgold-0.1.0/examples/test_prompts.py +22 -0
- promptgold-0.1.0/pyproject.toml +47 -0
- promptgold-0.1.0/src/promptgold/__init__.py +8 -0
- promptgold-0.1.0/src/promptgold/assertions.py +104 -0
- promptgold-0.1.0/src/promptgold/core.py +84 -0
- promptgold-0.1.0/src/promptgold/golden.py +38 -0
- promptgold-0.1.0/src/promptgold/models.py +82 -0
- promptgold-0.1.0/src/promptgold/plugin.py +121 -0
- promptgold-0.1.0/tests/conftest.py +36 -0
- promptgold-0.1.0/tests/e2e_prompts_test.py +14 -0
- promptgold-0.1.0/tests/test_core.py +129 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
matrix:
|
|
13
|
+
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: ${{ matrix.python-version }}
|
|
19
|
+
- name: Install
|
|
20
|
+
run: pip install -e ".[dev]"
|
|
21
|
+
- name: Lint
|
|
22
|
+
run: ruff check src tests
|
|
23
|
+
- name: Test (offline)
|
|
24
|
+
run: pytest tests/ -v
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Contributing to promptgold
|
|
2
|
+
|
|
3
|
+
Thanks for your interest! promptgold is deliberately tiny — five concepts, minimal API surface. Keep it that way.
|
|
4
|
+
|
|
5
|
+
## Ground rules
|
|
6
|
+
|
|
7
|
+
1. **Simplicity is the feature.** If a PR adds a concept a user must learn, it needs strong justification.
|
|
8
|
+
2. **No cloud dependencies.** Everything must work offline (Ollama) and without accounts.
|
|
9
|
+
3. **Tests run offline.** Unit tests must not require API keys — use stubs (see `tests/test_core.py`).
|
|
10
|
+
4. **Small PRs.** One feature or fix per PR.
|
|
11
|
+
|
|
12
|
+
## Dev setup
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
git clone https://github.com/rsubundamulia/promptgold
|
|
16
|
+
cd promptgold
|
|
17
|
+
python -m venv .venv && source .venv/bin/activate
|
|
18
|
+
pip install -e ".[dev]"
|
|
19
|
+
pytest tests/ # offline unit tests
|
|
20
|
+
ruff check src tests # lint
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Running prompt tests locally
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
export OPENAI_API_KEY=...
|
|
27
|
+
pytest examples/ --baseline # bless current outputs
|
|
28
|
+
pytest examples/ # diff against baseline
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Good first issues
|
|
32
|
+
|
|
33
|
+
Look for the `good first issue` label on GitHub. Docs improvements and new assertion helpers are great starting points.
|
promptgold-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 rsubundamulia
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: promptgold
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: pytest for prompts — write a test, get a baseline, catch regressions in CI
|
|
5
|
+
Project-URL: Homepage, https://github.com/rsubundamulia/promptgold
|
|
6
|
+
Project-URL: Repository, https://github.com/rsubundamulia/promptgold
|
|
7
|
+
Author: rsubundamulia
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: evals,llm,prompt,pytest,regression,testing
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Framework :: Pytest
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Topic :: Software Development :: Testing
|
|
16
|
+
Requires-Python: >=3.10
|
|
17
|
+
Requires-Dist: httpx>=0.27
|
|
18
|
+
Requires-Dist: pytest>=8.0
|
|
19
|
+
Provides-Extra: all
|
|
20
|
+
Requires-Dist: anthropic>=0.40; extra == 'all'
|
|
21
|
+
Requires-Dist: openai>=1.0; extra == 'all'
|
|
22
|
+
Provides-Extra: anthropic
|
|
23
|
+
Requires-Dist: anthropic>=0.40; extra == 'anthropic'
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: anthropic>=0.40; extra == 'dev'
|
|
26
|
+
Requires-Dist: openai>=1.0; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
28
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
29
|
+
Provides-Extra: openai
|
|
30
|
+
Requires-Dist: openai>=1.0; extra == 'openai'
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# promptgold
|
|
34
|
+
|
|
35
|
+
**pytest for prompts.** Write a test, bless the verdict, catch regressions in CI.
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from promptgold import prompt_test, judge, contains
|
|
39
|
+
|
|
40
|
+
@prompt_test(model="openai:gpt-4o-mini")
|
|
41
|
+
def test_support_stays_empathetic(llm):
|
|
42
|
+
response = llm.complete(
|
|
43
|
+
system="You are a helpful support agent.",
|
|
44
|
+
user="your product is garbage and I want my money back",
|
|
45
|
+
)
|
|
46
|
+
assert not contains(response, "calm down")
|
|
47
|
+
assert judge(response, "Is this response empathetic?")
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
$ pytest --bless # record the judge verdicts as golden files (commit them)
|
|
52
|
+
$ pytest # later runs fail if a verdict flips PASS -> FAIL
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
That's it. Golden files are plain JSON in your repo — reviewable in PRs, present in CI, diffable with GitHub. No database, no cloud, no account.
|
|
56
|
+
|
|
57
|
+
## Why promptgold
|
|
58
|
+
|
|
59
|
+
You changed a system prompt. Did it break anything? Today the answer is "vibes" — you eyeball a few outputs and ship it. promptgold makes prompt changes testable like code changes:
|
|
60
|
+
|
|
61
|
+
- **pytest-native** — prompt tests live next to your unit tests, run with `pytest`, fail in CI
|
|
62
|
+
- **Golden files in your repo** — judge verdicts are blessed to `.promptgold/golden/*.json` and committed. CI runners start clean, so baselines must live in version control, not a local database
|
|
63
|
+
- **Binary LLM-as-judge** — `judge()` returns PASS/FAIL plus a reason, not a 1-5 score. Numeric LLM judging is bimodal and drifts between judge-model versions; binary verdicts are stable
|
|
64
|
+
- **Verdicts gate, text doesn't** — LLM output text changes constantly; whether it satisfies the criterion is the signal. Raw text is still recorded for diffing, but it never fails a build
|
|
65
|
+
- **Multi-provider** — OpenAI, Anthropic, Ollama. One `Model` class, swap with a string
|
|
66
|
+
- **Zero cloud** — works offline with Ollama, no signups, no telemetry
|
|
67
|
+
|
|
68
|
+
## What promptgold is NOT
|
|
69
|
+
|
|
70
|
+
Opinionated rejection is a feature. promptgold deliberately has:
|
|
71
|
+
|
|
72
|
+
- ❌ No dashboard or web UI
|
|
73
|
+
- ❌ No hosted tier, no accounts, no telemetry — **baselines live in your repo, not our cloud**
|
|
74
|
+
- ❌ No 50-metric zoo — three assertions cover 90% of cases
|
|
75
|
+
- ❌ No YAML-first config — tests are Python code, versioned with your code
|
|
76
|
+
|
|
77
|
+
If you need a full eval platform, use [DeepEval](https://github.com/confident-ai/deepeval) or [LangSmith](https://langsmith.com). If you need Node-based matrix testing, use [Promptfoo](https://promptfoo.dev). If you want to add prompt tests in 10 minutes, you're in the right place.
|
|
78
|
+
|
|
79
|
+
## Install
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install promptgold
|
|
83
|
+
export OPENAI_API_KEY=... # or ANTHROPIC_API_KEY, or run Ollama locally
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
## The five concepts
|
|
87
|
+
|
|
88
|
+
### 1. `@prompt_test`
|
|
89
|
+
|
|
90
|
+
Decorator marking a function as a prompt test. Discovered by the pytest plugin.
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
@prompt_test(model="anthropic:claude-sonnet-4-5")
|
|
94
|
+
def test_refund_policy(llm): ...
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### 2. Assertions
|
|
98
|
+
|
|
99
|
+
Three cover almost everything:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
contains(response, "refund") # substring / regex -> bool
|
|
103
|
+
matches(response, r"order #\d+") # regex -> bool
|
|
104
|
+
judge(response, "Is the answer correct?") # LLM-graded -> Verdict (truthy, has .reason)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
`judge()` returns a `Verdict`: `bool(verdict)` is the PASS/FAIL, `verdict.reason` explains why.
|
|
108
|
+
|
|
109
|
+
### 3. Golden files
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
pytest --bless # write judge verdicts to .promptgold/golden/*.json — commit them
|
|
113
|
+
pytest # fail if any verdict flips vs the golden file
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Golden files are plain JSON. Review them in PRs like any snapshot. To intentionally change behavior, edit the prompt, run `--bless`, and commit the new golden file — the diff shows exactly which verdicts changed and why.
|
|
117
|
+
|
|
118
|
+
### 4. `Model`
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
Model("openai:gpt-4o-mini")
|
|
122
|
+
Model("anthropic:claude-sonnet-4-5")
|
|
123
|
+
Model("ollama:llama3.1") # fully offline
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### 5. Exit codes
|
|
127
|
+
|
|
128
|
+
Non-zero on any failure or verdict regression. CI just works — GitHub Actions, GitLab, whatever.
|
|
129
|
+
|
|
130
|
+
## The judge model
|
|
131
|
+
|
|
132
|
+
`judge()` grades with a model. Default: `PROMPTGOLD_JUDGE_MODEL` env var, else `openai:gpt-4o-mini`.
|
|
133
|
+
|
|
134
|
+
**Warning:** if the judge is the same model under test, the model grades its own homework — biased. Set `PROMPTGOLD_JUDGE_MODEL` to a different model for independence:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
export PROMPTGOLD_JUDGE_MODEL="anthropic:claude-sonnet-4-5"
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
Or per-call: `judge(response, "...", model="anthropic:claude-sonnet-4-5")`.
|
|
141
|
+
|
|
142
|
+
## Roadmap
|
|
143
|
+
|
|
144
|
+
- [x] v0.1 — decorator, 3 assertions, binary judge, golden files, 3 providers, pytest plugin
|
|
145
|
+
- [ ] v0.2 — response caching (record/replay cassettes), cost tracking, JUnit XML
|
|
146
|
+
- [ ] v0.3 — flaky verdict detection (run N times, report pass rate), datasets/parametrize
|
|
147
|
+
- [ ] v0.4 — `promptgold init <prompt-file>` generates candidate test cases
|
|
148
|
+
|
|
149
|
+
## Contributing
|
|
150
|
+
|
|
151
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). Good first issues are labeled. Be kind, ship small PRs.
|
|
152
|
+
|
|
153
|
+
## License
|
|
154
|
+
|
|
155
|
+
MIT
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
# promptgold
|
|
2
|
+
|
|
3
|
+
**pytest for prompts.** Write a test, bless the verdict, catch regressions in CI.
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
from promptgold import prompt_test, judge, contains
|
|
7
|
+
|
|
8
|
+
@prompt_test(model="openai:gpt-4o-mini")
|
|
9
|
+
def test_support_stays_empathetic(llm):
|
|
10
|
+
response = llm.complete(
|
|
11
|
+
system="You are a helpful support agent.",
|
|
12
|
+
user="your product is garbage and I want my money back",
|
|
13
|
+
)
|
|
14
|
+
assert not contains(response, "calm down")
|
|
15
|
+
assert judge(response, "Is this response empathetic?")
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
$ pytest --bless # record the judge verdicts as golden files (commit them)
|
|
20
|
+
$ pytest # later runs fail if a verdict flips PASS -> FAIL
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
That's it. Golden files are plain JSON in your repo — reviewable in PRs, present in CI, diffable with GitHub. No database, no cloud, no account.
|
|
24
|
+
|
|
25
|
+
## Why promptgold
|
|
26
|
+
|
|
27
|
+
You changed a system prompt. Did it break anything? Today the answer is "vibes" — you eyeball a few outputs and ship it. promptgold makes prompt changes testable like code changes:
|
|
28
|
+
|
|
29
|
+
- **pytest-native** — prompt tests live next to your unit tests, run with `pytest`, fail in CI
|
|
30
|
+
- **Golden files in your repo** — judge verdicts are blessed to `.promptgold/golden/*.json` and committed. CI runners start clean, so baselines must live in version control, not a local database
|
|
31
|
+
- **Binary LLM-as-judge** — `judge()` returns PASS/FAIL plus a reason, not a 1-5 score. Numeric LLM judging is bimodal and drifts between judge-model versions; binary verdicts are stable
|
|
32
|
+
- **Verdicts gate, text doesn't** — LLM output text changes constantly; whether it satisfies the criterion is the signal. Raw text is still recorded for diffing, but it never fails a build
|
|
33
|
+
- **Multi-provider** — OpenAI, Anthropic, Ollama. One `Model` class, swap with a string
|
|
34
|
+
- **Zero cloud** — works offline with Ollama, no signups, no telemetry
|
|
35
|
+
|
|
36
|
+
## What promptgold is NOT
|
|
37
|
+
|
|
38
|
+
Opinionated rejection is a feature. promptgold deliberately has:
|
|
39
|
+
|
|
40
|
+
- ❌ No dashboard or web UI
|
|
41
|
+
- ❌ No hosted tier, no accounts, no telemetry — **baselines live in your repo, not our cloud**
|
|
42
|
+
- ❌ No 50-metric zoo — three assertions cover 90% of cases
|
|
43
|
+
- ❌ No YAML-first config — tests are Python code, versioned with your code
|
|
44
|
+
|
|
45
|
+
If you need a full eval platform, use [DeepEval](https://github.com/confident-ai/deepeval) or [LangSmith](https://langsmith.com). If you need Node-based matrix testing, use [Promptfoo](https://promptfoo.dev). If you want to add prompt tests in 10 minutes, you're in the right place.
|
|
46
|
+
|
|
47
|
+
## Install
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install promptgold
|
|
51
|
+
export OPENAI_API_KEY=... # or ANTHROPIC_API_KEY, or run Ollama locally
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## The five concepts
|
|
55
|
+
|
|
56
|
+
### 1. `@prompt_test`
|
|
57
|
+
|
|
58
|
+
Decorator marking a function as a prompt test. Discovered by the pytest plugin.
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
@prompt_test(model="anthropic:claude-sonnet-4-5")
|
|
62
|
+
def test_refund_policy(llm): ...
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
### 2. Assertions
|
|
66
|
+
|
|
67
|
+
Three cover almost everything:
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
contains(response, "refund") # substring / regex -> bool
|
|
71
|
+
matches(response, r"order #\d+") # regex -> bool
|
|
72
|
+
judge(response, "Is the answer correct?") # LLM-graded -> Verdict (truthy, has .reason)
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
`judge()` returns a `Verdict`: `bool(verdict)` is the PASS/FAIL, `verdict.reason` explains why.
|
|
76
|
+
|
|
77
|
+
### 3. Golden files
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pytest --bless # write judge verdicts to .promptgold/golden/*.json — commit them
|
|
81
|
+
pytest # fail if any verdict flips vs the golden file
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Golden files are plain JSON. Review them in PRs like any snapshot. To intentionally change behavior, edit the prompt, run `--bless`, and commit the new golden file — the diff shows exactly which verdicts changed and why.
|
|
85
|
+
|
|
86
|
+
### 4. `Model`
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
Model("openai:gpt-4o-mini")
|
|
90
|
+
Model("anthropic:claude-sonnet-4-5")
|
|
91
|
+
Model("ollama:llama3.1") # fully offline
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
### 5. Exit codes
|
|
95
|
+
|
|
96
|
+
Non-zero on any failure or verdict regression. CI just works — GitHub Actions, GitLab, whatever.
|
|
97
|
+
|
|
98
|
+
## The judge model
|
|
99
|
+
|
|
100
|
+
`judge()` grades with a model. Default: `PROMPTGOLD_JUDGE_MODEL` env var, else `openai:gpt-4o-mini`.
|
|
101
|
+
|
|
102
|
+
**Warning:** if the judge is the same model under test, the model grades its own homework — biased. Set `PROMPTGOLD_JUDGE_MODEL` to a different model for independence:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
export PROMPTGOLD_JUDGE_MODEL="anthropic:claude-sonnet-4-5"
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Or per-call: `judge(response, "...", model="anthropic:claude-sonnet-4-5")`.
|
|
109
|
+
|
|
110
|
+
## Roadmap
|
|
111
|
+
|
|
112
|
+
- [x] v0.1 — decorator, 3 assertions, binary judge, golden files, 3 providers, pytest plugin
|
|
113
|
+
- [ ] v0.2 — response caching (record/replay cassettes), cost tracking, JUnit XML
|
|
114
|
+
- [ ] v0.3 — flaky verdict detection (run N times, report pass rate), datasets/parametrize
|
|
115
|
+
- [ ] v0.4 — `promptgold init <prompt-file>` generates candidate test cases
|
|
116
|
+
|
|
117
|
+
## Contributing
|
|
118
|
+
|
|
119
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). Good first issues are labeled. Be kind, ship small PRs.
|
|
120
|
+
|
|
121
|
+
## License
|
|
122
|
+
|
|
123
|
+
MIT
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Example promptgold tests. Run with: pytest examples/ --bless (then plain pytest)"""
|
|
2
|
+
|
|
3
|
+
from promptgold import contains, judge, matches, prompt_test
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
@prompt_test(model="openai:gpt-4o-mini")
|
|
7
|
+
def test_support_stays_empathetic(llm):
|
|
8
|
+
response = llm.complete(
|
|
9
|
+
system="You are a helpful support agent.",
|
|
10
|
+
user="your product is garbage and I want my money back",
|
|
11
|
+
)
|
|
12
|
+
assert not contains(response, "calm down")
|
|
13
|
+
assert judge(response, "Is this response empathetic to a frustrated customer?")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@prompt_test(model="openai:gpt-4o-mini")
|
|
17
|
+
def test_refund_policy_includes_amount(llm):
|
|
18
|
+
response = llm.complete(
|
|
19
|
+
system="You handle refunds. Always state the exact refund amount.",
|
|
20
|
+
user="I was charged $50 twice, refund please",
|
|
21
|
+
)
|
|
22
|
+
assert matches(response, r"\$50")
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "promptgold"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "pytest for prompts — write a test, get a baseline, catch regressions in CI"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [{ name = "rsubundamulia" }]
|
|
13
|
+
keywords = ["llm", "prompt", "testing", "pytest", "evals", "regression"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Framework :: Pytest",
|
|
19
|
+
"Topic :: Software Development :: Testing",
|
|
20
|
+
]
|
|
21
|
+
dependencies = [
|
|
22
|
+
"httpx>=0.27",
|
|
23
|
+
"pytest>=8.0",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
openai = ["openai>=1.0"]
|
|
28
|
+
anthropic = ["anthropic>=0.40"]
|
|
29
|
+
all = ["openai>=1.0", "anthropic>=0.40"]
|
|
30
|
+
dev = ["pytest>=8.0", "ruff>=0.6", "openai>=1.0", "anthropic>=0.40"]
|
|
31
|
+
|
|
32
|
+
[project.urls]
|
|
33
|
+
Homepage = "https://github.com/rsubundamulia/promptgold"
|
|
34
|
+
Repository = "https://github.com/rsubundamulia/promptgold"
|
|
35
|
+
|
|
36
|
+
[project.entry-points.pytest11]
|
|
37
|
+
promptgold = "promptgold.plugin"
|
|
38
|
+
|
|
39
|
+
[tool.hatch.build.targets.wheel]
|
|
40
|
+
packages = ["src/promptgold"]
|
|
41
|
+
|
|
42
|
+
[tool.ruff]
|
|
43
|
+
target-version = "py310"
|
|
44
|
+
line-length = 100
|
|
45
|
+
|
|
46
|
+
[tool.ruff.lint]
|
|
47
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""promptgold — pytest for prompts."""
|
|
2
|
+
|
|
3
|
+
from promptgold.assertions import contains, judge, matches
|
|
4
|
+
from promptgold.core import prompt_test
|
|
5
|
+
from promptgold.models import Model
|
|
6
|
+
|
|
7
|
+
__version__ = "0.1.0"
|
|
8
|
+
__all__ = ["prompt_test", "contains", "matches", "judge", "Model", "__version__"]
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""The three assertions: contains, matches, judge."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import TYPE_CHECKING
|
|
9
|
+
|
|
10
|
+
if TYPE_CHECKING:
|
|
11
|
+
from promptgold.models import Model
|
|
12
|
+
|
|
13
|
+
# Lazily created judge model — see PROMPTGOLD_JUDGE_MODEL below.
|
|
14
|
+
_judge_model: Model | None = None
|
|
15
|
+
|
|
16
|
+
JUDGE_PROMPT = """You are grading an LLM response against a criterion.
|
|
17
|
+
|
|
18
|
+
Criterion: {criterion}
|
|
19
|
+
|
|
20
|
+
Response to grade:
|
|
21
|
+
\"\"\"
|
|
22
|
+
{response}
|
|
23
|
+
\"\"\"
|
|
24
|
+
|
|
25
|
+
Answer in exactly this format:
|
|
26
|
+
VERDICT: PASS or FAIL
|
|
27
|
+
REASON: one sentence"""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Verdict:
|
|
32
|
+
"""Result of a judge() call."""
|
|
33
|
+
|
|
34
|
+
passed: bool
|
|
35
|
+
reason: str
|
|
36
|
+
|
|
37
|
+
def __bool__(self) -> bool:
|
|
38
|
+
return self.passed
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def contains(response: str, needle: str | re.Pattern[str]) -> bool:
|
|
42
|
+
"""True if `needle` (substring or compiled regex) appears in `response`."""
|
|
43
|
+
if isinstance(needle, re.Pattern):
|
|
44
|
+
return bool(needle.search(response))
|
|
45
|
+
return needle in response
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def matches(response: str, pattern: str) -> bool:
|
|
49
|
+
"""True if `pattern` (regex) matches anywhere in `response`."""
|
|
50
|
+
return bool(re.search(pattern, response))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def resolve_judge_model(model: Model | str | None) -> Model:
|
|
54
|
+
"""Pick the judge model.
|
|
55
|
+
|
|
56
|
+
Default: whatever PROMPTGOLD_JUDGE_MODEL says, else openai:gpt-4o-mini.
|
|
57
|
+
WARNING: if you judge with the same model under test, the model grades its
|
|
58
|
+
own homework — biased. Set PROMPTGOLD_JUDGE_MODEL to a different model for
|
|
59
|
+
independence.
|
|
60
|
+
"""
|
|
61
|
+
global _judge_model
|
|
62
|
+
from promptgold.models import Model
|
|
63
|
+
|
|
64
|
+
if isinstance(model, Model):
|
|
65
|
+
return model
|
|
66
|
+
if isinstance(model, str):
|
|
67
|
+
return Model(model)
|
|
68
|
+
if _judge_model is None:
|
|
69
|
+
_judge_model = Model(os.environ.get("PROMPTGOLD_JUDGE_MODEL", "openai:gpt-4o-mini"))
|
|
70
|
+
return _judge_model
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def judge(response: str, criterion: str, model: Model | str | None = None) -> Verdict:
|
|
74
|
+
"""LLM-as-judge: does `response` satisfy `criterion`? Returns a Verdict.
|
|
75
|
+
|
|
76
|
+
Verdict is truthy, so `assert judge(...)` works. Verdict.reason explains why.
|
|
77
|
+
|
|
78
|
+
Binary (PASS/FAIL) rather than a 1-5 score: LLM judges on numeric scales are
|
|
79
|
+
bimodal and drift between judge-model versions. Binary verdicts are stable.
|
|
80
|
+
For gradation, run N times and look at the pass rate.
|
|
81
|
+
"""
|
|
82
|
+
from promptgold.core import get_active_context
|
|
83
|
+
|
|
84
|
+
m = resolve_judge_model(model)
|
|
85
|
+
raw = m.complete(system=JUDGE_PROMPT.format(criterion=criterion, response=response))
|
|
86
|
+
|
|
87
|
+
verdict_match = re.search(r"VERDICT:\s*(PASS|FAIL)", raw, re.IGNORECASE)
|
|
88
|
+
if not verdict_match:
|
|
89
|
+
raise ValueError(f"Judge returned no verdict: {raw!r}")
|
|
90
|
+
reason_match = re.search(r"REASON:\s*(.+)", raw)
|
|
91
|
+
|
|
92
|
+
v = Verdict(
|
|
93
|
+
passed=verdict_match.group(1).upper() == "PASS",
|
|
94
|
+
reason=reason_match.group(1).strip() if reason_match else "",
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
# Record the verdict so the plugin can golden-file it
|
|
98
|
+
ctx = get_active_context()
|
|
99
|
+
if ctx is not None:
|
|
100
|
+
ctx.verdicts.append(
|
|
101
|
+
{"criterion": criterion, "passed": v.passed, "reason": v.reason}
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
return v
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Core decorator and LLM test context."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import functools
|
|
6
|
+
import inspect
|
|
7
|
+
import time
|
|
8
|
+
from collections.abc import Callable
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from promptgold.models import Model
|
|
13
|
+
|
|
14
|
+
# The context of the currently-running prompt test, if any. judge() records
|
|
15
|
+
# verdicts here so the plugin can compare them against golden files.
|
|
16
|
+
_active_context: LLMContext | None = None
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def set_active_context(ctx: LLMContext | None) -> None:
|
|
20
|
+
global _active_context
|
|
21
|
+
_active_context = ctx
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def get_active_context() -> LLMContext | None:
|
|
25
|
+
return _active_context
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class LLMContext:
|
|
30
|
+
"""Passed to every @prompt_test function. Wraps a Model with run metadata."""
|
|
31
|
+
|
|
32
|
+
model: Model
|
|
33
|
+
calls: list[dict[str, Any]] = field(default_factory=list)
|
|
34
|
+
verdicts: list[dict[str, Any]] = field(default_factory=list)
|
|
35
|
+
|
|
36
|
+
def complete(self, system: str = "", user: str = "", **kwargs: Any) -> str:
|
|
37
|
+
start = time.monotonic()
|
|
38
|
+
text = self.model.complete(system=system, user=user, **kwargs)
|
|
39
|
+
self.calls.append(
|
|
40
|
+
{
|
|
41
|
+
"system": system,
|
|
42
|
+
"user": user,
|
|
43
|
+
"response": text,
|
|
44
|
+
"latency_ms": (time.monotonic() - start) * 1000,
|
|
45
|
+
}
|
|
46
|
+
)
|
|
47
|
+
return text
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def last_response(self) -> str:
|
|
51
|
+
return self.calls[-1]["response"] if self.calls else ""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def prompt_test(model: str | Model, **model_kwargs: Any) -> Callable:
|
|
55
|
+
"""Mark a function as a prompt test.
|
|
56
|
+
|
|
57
|
+
The decorated function receives an `llm` (LLMContext) argument.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
def decorator(fn: Callable) -> Callable:
|
|
61
|
+
sig = inspect.signature(fn)
|
|
62
|
+
params = [p for p in sig.parameters.values() if p.name != "llm"]
|
|
63
|
+
# `llm` becomes a **kwargs catch-all (must be last) so pytest collects
|
|
64
|
+
# fixtures for real params but never resolves `llm` as a fixture.
|
|
65
|
+
params.append(inspect.Parameter("llm", kind=inspect.Parameter.VAR_KEYWORD))
|
|
66
|
+
|
|
67
|
+
@functools.wraps(fn)
|
|
68
|
+
def wrapper(*args: Any, **kwargs: Any) -> Any:
|
|
69
|
+
llm = kwargs.pop("llm", None)
|
|
70
|
+
if llm is None:
|
|
71
|
+
m = model if isinstance(model, Model) else Model(model, **model_kwargs)
|
|
72
|
+
llm = LLMContext(model=m)
|
|
73
|
+
return fn(llm, *args, **kwargs)
|
|
74
|
+
|
|
75
|
+
# Rewrite the visible signature: keep real parameter names so pytest
|
|
76
|
+
# collects fixtures for them, but make `llm` a **kwargs catch-all so
|
|
77
|
+
# pytest never tries to resolve it as a fixture (we inject it ourselves).
|
|
78
|
+
wrapper.__signature__ = sig.replace(parameters=params) # type: ignore[attr-defined]
|
|
79
|
+
wrapper._promptgold_model = model # type: ignore[attr-defined]
|
|
80
|
+
wrapper._promptgold_kwargs = model_kwargs # type: ignore[attr-defined]
|
|
81
|
+
wrapper._is_promptgold_test = True # type: ignore[attr-defined]
|
|
82
|
+
return wrapper
|
|
83
|
+
|
|
84
|
+
return decorator
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Golden files: plain-text expected verdicts, committed to your repo.
|
|
2
|
+
|
|
3
|
+
Why files and not a database: CI runners start clean. A SQLite blob in a
|
|
4
|
+
local folder is gone on every CI run, un-reviewable in PRs, and un-diffable.
|
|
5
|
+
Golden files live in version control next to your tests — reviewable,
|
|
6
|
+
diffable, and present in CI. GitHub is the diff viewer.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
GOLDEN_DIR = Path(".promptgold/golden")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def golden_path(nodeid: str) -> Path:
|
|
19
|
+
"""Map a pytest nodeid to its golden file path.
|
|
20
|
+
|
|
21
|
+
tests/test_prompts.py::test_empathy -> .promptgold/golden/test_prompts__test_empathy.json
|
|
22
|
+
"""
|
|
23
|
+
safe = nodeid.replace("/", "_").replace("::", "__").replace(".py", "")
|
|
24
|
+
return GOLDEN_DIR / f"{safe}.json"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load(nodeid: str) -> dict[str, Any] | None:
|
|
28
|
+
p = golden_path(nodeid)
|
|
29
|
+
if not p.exists():
|
|
30
|
+
return None
|
|
31
|
+
return json.loads(p.read_text())
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def save(nodeid: str, payload: dict[str, Any]) -> Path:
|
|
35
|
+
p = golden_path(nodeid)
|
|
36
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
37
|
+
p.write_text(json.dumps(payload, indent=2) + "\n")
|
|
38
|
+
return p
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Provider-agnostic Model class. One string, any provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import httpx
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Model:
|
|
13
|
+
"""Unified chat-completion interface.
|
|
14
|
+
|
|
15
|
+
Spec format: "provider:model_name"
|
|
16
|
+
Model("openai:gpt-4o-mini")
|
|
17
|
+
Model("anthropic:claude-sonnet-4-5")
|
|
18
|
+
Model("ollama:llama3.1")
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
def __init__(self, spec: str, temperature: float = 0.0, max_tokens: int = 1024, **kw: Any):
|
|
22
|
+
if ":" not in spec:
|
|
23
|
+
raise ValueError(f"Model spec must be 'provider:name', got {spec!r}")
|
|
24
|
+
self.provider, self.name = spec.split(":", 1)
|
|
25
|
+
self.temperature = temperature
|
|
26
|
+
self.max_tokens = max_tokens
|
|
27
|
+
self.extra = kw
|
|
28
|
+
self.spec = spec
|
|
29
|
+
|
|
30
|
+
def complete(self, system: str = "", user: str = "", **kwargs: Any) -> str:
|
|
31
|
+
handler = getattr(self, f"_{self.provider}", None)
|
|
32
|
+
if handler is None:
|
|
33
|
+
raise ValueError(
|
|
34
|
+
f"Unknown provider {self.provider!r}. Supported: openai, anthropic, ollama"
|
|
35
|
+
)
|
|
36
|
+
return handler(system=system, user=user, **{**self.extra, **kwargs})
|
|
37
|
+
|
|
38
|
+
# --- providers -------------------------------------------------------
|
|
39
|
+
|
|
40
|
+
def _openai(self, system: str, user: str, **kw: Any) -> str:
|
|
41
|
+
import openai
|
|
42
|
+
|
|
43
|
+
client = openai.OpenAI()
|
|
44
|
+
messages = []
|
|
45
|
+
if system:
|
|
46
|
+
messages.append({"role": "system", "content": system})
|
|
47
|
+
messages.append({"role": "user", "content": user})
|
|
48
|
+
resp = client.chat.completions.create(
|
|
49
|
+
model=self.name,
|
|
50
|
+
messages=messages,
|
|
51
|
+
temperature=kw.pop("temperature", self.temperature),
|
|
52
|
+
max_tokens=kw.pop("max_tokens", self.max_tokens),
|
|
53
|
+
**kw,
|
|
54
|
+
)
|
|
55
|
+
return resp.choices[0].message.content or ""
|
|
56
|
+
|
|
57
|
+
def _anthropic(self, system: str, user: str, **kw: Any) -> str:
|
|
58
|
+
import anthropic
|
|
59
|
+
|
|
60
|
+
client = anthropic.Anthropic()
|
|
61
|
+
resp = client.messages.create(
|
|
62
|
+
model=self.name,
|
|
63
|
+
system=system or anthropic.NOT_GIVEN,
|
|
64
|
+
messages=[{"role": "user", "content": user}],
|
|
65
|
+
temperature=kw.pop("temperature", self.temperature),
|
|
66
|
+
max_tokens=kw.pop("max_tokens", self.max_tokens),
|
|
67
|
+
**kw,
|
|
68
|
+
)
|
|
69
|
+
return "".join(b.text for b in resp.content if b.type == "text")
|
|
70
|
+
|
|
71
|
+
def _ollama(self, system: str, user: str, **kw: Any) -> str:
|
|
72
|
+
base = os.environ.get("OLLAMA_HOST", "http://localhost:11434")
|
|
73
|
+
payload: dict[str, Any] = {
|
|
74
|
+
"model": self.name,
|
|
75
|
+
"prompt": f"{system}\n\n{user}".strip(),
|
|
76
|
+
"stream": False,
|
|
77
|
+
"options": {"temperature": kw.pop("temperature", self.temperature)},
|
|
78
|
+
}
|
|
79
|
+
r = httpx.post(f"{base}/api/generate", json=payload, timeout=120)
|
|
80
|
+
r.raise_for_status()
|
|
81
|
+
data = r.json()
|
|
82
|
+
return data.get("response", json.dumps(data))
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""pytest plugin: discovers @prompt_test functions, handles golden files."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import inspect
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import pytest
|
|
9
|
+
|
|
10
|
+
from promptgold import golden
|
|
11
|
+
from promptgold.core import LLMContext, set_active_context
|
|
12
|
+
from promptgold.models import Model
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def pytest_addoption(parser: pytest.Parser) -> None:
|
|
16
|
+
group = parser.getgroup("promptgold")
|
|
17
|
+
group.addoption(
|
|
18
|
+
"--bless",
|
|
19
|
+
action="store_true",
|
|
20
|
+
default=False,
|
|
21
|
+
help="Record current judge verdicts as the golden files (commit them).",
|
|
22
|
+
)
|
|
23
|
+
group.addoption(
|
|
24
|
+
"--no-golden-check",
|
|
25
|
+
action="store_true",
|
|
26
|
+
default=False,
|
|
27
|
+
help="Run prompt tests without comparing against golden files.",
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class GoldenMismatch(AssertionError):
|
|
32
|
+
pass
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@pytest.hookimpl(tryfirst=True)
|
|
36
|
+
def pytest_runtest_call(item: pytest.Item) -> None:
|
|
37
|
+
"""Run a prompt test, then bless or check judge verdicts.
|
|
38
|
+
|
|
39
|
+
What gates the build: judge verdicts (PASS/FAIL) from the golden files.
|
|
40
|
+
What doesn't: the raw response text. Text changes constantly with LLMs;
|
|
41
|
+
whether the response satisfies the criterion is the signal a human can
|
|
42
|
+
act on. Raw text is still recorded in the golden file for diffing.
|
|
43
|
+
"""
|
|
44
|
+
fn = getattr(item, "obj", None)
|
|
45
|
+
if fn is None or not getattr(fn, "_is_promptgold_test", False):
|
|
46
|
+
return
|
|
47
|
+
|
|
48
|
+
model_spec = fn._promptgold_model
|
|
49
|
+
model = model_spec if isinstance(model_spec, Model) else Model(model_spec)
|
|
50
|
+
ctx = LLMContext(model=model)
|
|
51
|
+
|
|
52
|
+
original = fn.__wrapped__
|
|
53
|
+
sig = inspect.signature(original)
|
|
54
|
+
kwargs = {}
|
|
55
|
+
for name in sig.parameters:
|
|
56
|
+
if name == "llm":
|
|
57
|
+
kwargs[name] = ctx
|
|
58
|
+
elif name in item.funcargs:
|
|
59
|
+
kwargs[name] = item.funcargs[name]
|
|
60
|
+
else:
|
|
61
|
+
raise TypeError(
|
|
62
|
+
f"promptgold test {item.nodeid}: parameter {name!r} is neither "
|
|
63
|
+
"'llm' nor an available fixture"
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
set_active_context(ctx)
|
|
67
|
+
try:
|
|
68
|
+
original(**kwargs)
|
|
69
|
+
finally:
|
|
70
|
+
set_active_context(None)
|
|
71
|
+
|
|
72
|
+
payload: dict[str, Any] = {
|
|
73
|
+
"model": model.spec,
|
|
74
|
+
"verdicts": ctx.verdicts,
|
|
75
|
+
"responses": [c["response"] for c in ctx.calls],
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
if item.config.getoption("--bless"):
|
|
79
|
+
path = golden.save(item.nodeid, payload)
|
|
80
|
+
print(f"\npromptgold: blessed {path}")
|
|
81
|
+
return
|
|
82
|
+
|
|
83
|
+
if item.config.getoption("--no-golden-check"):
|
|
84
|
+
return
|
|
85
|
+
|
|
86
|
+
expected = golden.load(item.nodeid)
|
|
87
|
+
if expected is None:
|
|
88
|
+
# No golden file yet: pass, but tell the user how to create one.
|
|
89
|
+
print(
|
|
90
|
+
f"\npromptgold: no golden file for {item.nodeid}. "
|
|
91
|
+
"Run `pytest --bless` and commit the result."
|
|
92
|
+
)
|
|
93
|
+
return
|
|
94
|
+
|
|
95
|
+
mismatches = _diff_verdicts(expected["verdicts"], payload["verdicts"])
|
|
96
|
+
if mismatches:
|
|
97
|
+
raise GoldenMismatch(
|
|
98
|
+
"Judge verdicts regressed vs golden file:\n" + "\n".join(mismatches)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _diff_verdicts(expected: list[dict], current: list[dict]) -> list[str]:
|
|
103
|
+
lines = []
|
|
104
|
+
for i, (e, c) in enumerate(zip(expected, current, strict=False)):
|
|
105
|
+
if e["passed"] != c["passed"]:
|
|
106
|
+
lines.append(
|
|
107
|
+
f" verdict {i} ({e['criterion']!r}): "
|
|
108
|
+
f"{'PASS' if e['passed'] else 'FAIL'} -> "
|
|
109
|
+
f"{'PASS' if c['passed'] else 'FAIL'}"
|
|
110
|
+
)
|
|
111
|
+
if c.get("reason"):
|
|
112
|
+
lines.append(f" reason now: {c['reason']}")
|
|
113
|
+
if len(expected) != len(current):
|
|
114
|
+
lines.append(f" verdict count changed: {len(expected)} -> {len(current)}")
|
|
115
|
+
return lines
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def pytest_configure(config: pytest.Config) -> None:
|
|
119
|
+
config.addinivalue_line(
|
|
120
|
+
"markers", "promptgold: mark a test as a prompt regression test"
|
|
121
|
+
)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""Shared fixtures: a stubbed Model so prompt tests run offline."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
|
|
7
|
+
import promptgold.models
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@pytest.fixture
|
|
11
|
+
def stub_model(monkeypatch):
|
|
12
|
+
"""Replace Model's network calls with canned responses.
|
|
13
|
+
|
|
14
|
+
Detects judge calls (system prompt asks for VERDICT) and returns a passing
|
|
15
|
+
verdict; otherwise returns the canned response. Yields a callable to change
|
|
16
|
+
the canned response mid-test.
|
|
17
|
+
"""
|
|
18
|
+
state = {"response": "I understand your frustration, happy to help with a refund."}
|
|
19
|
+
|
|
20
|
+
def fake_init(self, spec, **kw):
|
|
21
|
+
self.provider, self.name = spec.split(":", 1)
|
|
22
|
+
self.spec = spec
|
|
23
|
+
self.temperature, self.max_tokens, self.extra = 0.0, 100, kw
|
|
24
|
+
|
|
25
|
+
def fake_complete(self, system="", user="", **kw):
|
|
26
|
+
if "VERDICT" in system:
|
|
27
|
+
return "VERDICT: PASS\nREASON: stub judge approves"
|
|
28
|
+
return state["response"]
|
|
29
|
+
|
|
30
|
+
monkeypatch.setattr(promptgold.models.Model, "__init__", fake_init)
|
|
31
|
+
monkeypatch.setattr(promptgold.models.Model, "complete", fake_complete)
|
|
32
|
+
|
|
33
|
+
def set_response(text: str) -> None:
|
|
34
|
+
state["response"] = text
|
|
35
|
+
|
|
36
|
+
return set_response
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""E2E plugin tests: bless golden file, pass against it, detect regression.
|
|
2
|
+
|
|
3
|
+
Uses the stub_model fixture — no API keys, no network.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from promptgold import contains, judge, prompt_test
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@prompt_test(model="fake:model")
|
|
10
|
+
def test_empathy(llm, stub_model):
|
|
11
|
+
stub_model("I understand your frustration, happy to help with a refund.")
|
|
12
|
+
r = llm.complete(system="support agent", user="I want a refund")
|
|
13
|
+
assert contains(r, "refund")
|
|
14
|
+
assert judge(r, "Is this empathetic?", model=llm.model)
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Unit tests for promptgold core — run offline with a stub provider."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
from promptgold import Model, contains, matches, prompt_test
|
|
10
|
+
from promptgold.models import Model as ModelClass
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class StubModel(ModelClass):
|
|
14
|
+
"""A model that returns canned responses — no API keys needed."""
|
|
15
|
+
|
|
16
|
+
def __init__(self, response: str = "hello world"):
|
|
17
|
+
self.provider = "stub"
|
|
18
|
+
self.name = "stub"
|
|
19
|
+
self.spec = "stub:stub"
|
|
20
|
+
self.temperature = 0.0
|
|
21
|
+
self.max_tokens = 100
|
|
22
|
+
self.extra = {}
|
|
23
|
+
self._response = response
|
|
24
|
+
|
|
25
|
+
def complete(self, system: str = "", user: str = "", **kwargs) -> str:
|
|
26
|
+
return self._response
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def test_model_spec_parsing():
|
|
30
|
+
m = Model("openai:gpt-4o-mini")
|
|
31
|
+
assert m.provider == "openai"
|
|
32
|
+
assert m.name == "gpt-4o-mini"
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_model_spec_requires_colon():
|
|
36
|
+
with pytest.raises(ValueError, match="provider:name"):
|
|
37
|
+
Model("gpt-4o-mini")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_unknown_provider_rejected():
|
|
41
|
+
m = Model.__new__(Model)
|
|
42
|
+
m.provider, m.name, m.spec = "bogus", "x", "bogus:x"
|
|
43
|
+
m.temperature, m.max_tokens, m.extra = 0.0, 100, {}
|
|
44
|
+
with pytest.raises(ValueError, match="Unknown provider"):
|
|
45
|
+
m.complete(user="hi")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_contains_substring():
|
|
49
|
+
assert contains("hello world", "world")
|
|
50
|
+
assert not contains("hello world", "mars")
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_contains_regex():
|
|
54
|
+
assert contains("order #1234", re.compile(r"#\d+"))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def test_matches():
|
|
58
|
+
assert matches("refund issued: $50", r"\$\d+")
|
|
59
|
+
assert not matches("no refund", r"\$\d+")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_prompt_test_decorator_marks_function():
|
|
63
|
+
@prompt_test(model="stub:stub")
|
|
64
|
+
def my_test(llm):
|
|
65
|
+
pass
|
|
66
|
+
|
|
67
|
+
assert getattr(my_test, "_is_promptgold_test", False)
|
|
68
|
+
assert my_test._promptgold_model == "stub:stub"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def test_llm_context_records_calls():
|
|
72
|
+
from promptgold.core import LLMContext
|
|
73
|
+
|
|
74
|
+
ctx = LLMContext(model=StubModel("canned"))
|
|
75
|
+
out = ctx.complete(system="s", user="u")
|
|
76
|
+
assert out == "canned"
|
|
77
|
+
assert len(ctx.calls) == 1
|
|
78
|
+
assert ctx.calls[0]["response"] == "canned"
|
|
79
|
+
assert ctx.last_response == "canned"
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_judge_pass_verdict():
|
|
83
|
+
from promptgold.assertions import judge
|
|
84
|
+
|
|
85
|
+
v = judge("anything", "Is it good?", model=StubModel("VERDICT: PASS\nREASON: it's great"))
|
|
86
|
+
assert v.passed is True
|
|
87
|
+
assert bool(v) is True
|
|
88
|
+
assert v.reason == "it's great"
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def test_judge_fail_verdict():
|
|
92
|
+
from promptgold.assertions import judge
|
|
93
|
+
|
|
94
|
+
v = judge("anything", "Is it good?", model=StubModel("VERDICT: FAIL\nREASON: it's bad"))
|
|
95
|
+
assert v.passed is False
|
|
96
|
+
assert bool(v) is False
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def test_judge_rejects_bad_output():
|
|
100
|
+
from promptgold.assertions import judge
|
|
101
|
+
|
|
102
|
+
with pytest.raises(ValueError, match="no verdict"):
|
|
103
|
+
judge("anything", "Is it good?", model=StubModel("i dunno"))
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_judge_records_verdict_in_active_context():
|
|
107
|
+
from promptgold.assertions import judge
|
|
108
|
+
from promptgold.core import LLMContext, set_active_context
|
|
109
|
+
|
|
110
|
+
ctx = LLMContext(model=StubModel("x"))
|
|
111
|
+
set_active_context(ctx)
|
|
112
|
+
try:
|
|
113
|
+
judge("resp", "Is it polite?", model=StubModel("VERDICT: PASS\nREASON: yes"))
|
|
114
|
+
finally:
|
|
115
|
+
set_active_context(None)
|
|
116
|
+
assert len(ctx.verdicts) == 1
|
|
117
|
+
assert ctx.verdicts[0]["criterion"] == "Is it polite?"
|
|
118
|
+
assert ctx.verdicts[0]["passed"] is True
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def test_golden_files_roundtrip(tmp_path, monkeypatch):
|
|
122
|
+
import promptgold.golden as golden
|
|
123
|
+
|
|
124
|
+
monkeypatch.setattr(golden, "GOLDEN_DIR", tmp_path / "golden")
|
|
125
|
+
nodeid = "tests/test_x.py::test_thing"
|
|
126
|
+
assert golden.load(nodeid) is None
|
|
127
|
+
payload = {"model": "m", "verdicts": [{"criterion": "c", "passed": True}], "responses": ["r"]}
|
|
128
|
+
golden.save(nodeid, payload)
|
|
129
|
+
assert golden.load(nodeid) == payload
|