promptseal 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- promptseal-0.2.0/LICENSE +21 -0
- promptseal-0.2.0/PKG-INFO +255 -0
- promptseal-0.2.0/README.md +219 -0
- promptseal-0.2.0/pyproject.toml +64 -0
- promptseal-0.2.0/setup.cfg +4 -0
- promptseal-0.2.0/src/promptseal/__init__.py +5 -0
- promptseal-0.2.0/src/promptseal/__main__.py +3 -0
- promptseal-0.2.0/src/promptseal/_version.py +1 -0
- promptseal-0.2.0/src/promptseal/assertions.py +212 -0
- promptseal-0.2.0/src/promptseal/cli.py +389 -0
- promptseal-0.2.0/src/promptseal/config.py +111 -0
- promptseal-0.2.0/src/promptseal/diff.py +61 -0
- promptseal-0.2.0/src/promptseal/init_templates.py +53 -0
- promptseal-0.2.0/src/promptseal/models.py +88 -0
- promptseal-0.2.0/src/promptseal/providers.py +166 -0
- promptseal-0.2.0/src/promptseal/record.py +281 -0
- promptseal-0.2.0/src/promptseal/report.py +193 -0
- promptseal-0.2.0/src/promptseal/report_html.py +225 -0
- promptseal-0.2.0/src/promptseal/runner.py +201 -0
- promptseal-0.2.0/src/promptseal/storage.py +96 -0
- promptseal-0.2.0/src/promptseal.egg-info/PKG-INFO +255 -0
- promptseal-0.2.0/src/promptseal.egg-info/SOURCES.txt +29 -0
- promptseal-0.2.0/src/promptseal.egg-info/dependency_links.txt +1 -0
- promptseal-0.2.0/src/promptseal.egg-info/entry_points.txt +2 -0
- promptseal-0.2.0/src/promptseal.egg-info/requires.txt +11 -0
- promptseal-0.2.0/src/promptseal.egg-info/top_level.txt +1 -0
- promptseal-0.2.0/tests/test_assertions.py +112 -0
- promptseal-0.2.0/tests/test_cli.py +125 -0
- promptseal-0.2.0/tests/test_diff.py +69 -0
- promptseal-0.2.0/tests/test_record.py +139 -0
- promptseal-0.2.0/tests/test_runner.py +102 -0
promptseal-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 MohammadReza Shabani
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: promptseal
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: π¦ Regression testing for prompts, agents, and models. Know what breaks before you switch.
|
|
5
|
+
Author: Arsin Shaabani
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/ArsinShaabani/promptseal
|
|
8
|
+
Project-URL: Issues, https://github.com/ArsinShaabani/promptseal/issues
|
|
9
|
+
Project-URL: Roadmap, https://github.com/ArsinShaabani/promptseal/blob/main/ROADMAP.md
|
|
10
|
+
Keywords: llm,evals,prompt,regression,testing,ci,ai-agents,model-comparison
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
21
|
+
Classifier: Topic :: Software Development :: Testing
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Requires-Dist: typer>=0.12
|
|
26
|
+
Requires-Dist: rich>=13.0
|
|
27
|
+
Requires-Dist: httpx>=0.27
|
|
28
|
+
Requires-Dist: pyyaml>=6.0
|
|
29
|
+
Requires-Dist: pydantic>=2.5
|
|
30
|
+
Requires-Dist: jinja2>=3.1
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest-cov>=4.1; extra == "dev"
|
|
34
|
+
Requires-Dist: ruff>=0.4; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# π¦ PromptSeal
|
|
38
|
+
|
|
39
|
+

|
|
40
|
+
|
|
41
|
+
**Regression testing for prompts, agents, and models. Know what breaks *before* you switch.**
|
|
42
|
+
|
|
43
|
+
π **Read this in Persian (ΩΨ§Ψ±Ψ³Ϋ): [README.fa.md](README.fa.md)** Β·
|
|
44
|
+
π **Full tutorial: [TUTORIAL.md](TUTORIAL.md) | [Ψ’Ω
ΩΨ²Ψ΄ Ϊ©Ψ§Ω
Ω ΩΨ§Ψ±Ψ³Ϋ](TUTORIAL.fa.md)**
|
|
45
|
+
|
|
46
|
+
You changed one word in your system prompt. Or swapped `gpt-4o` for that shiny new
|
|
47
|
+
open-weights model. Did you just break your app? **Nobody knows β until your users do.**
|
|
48
|
+
|
|
49
|
+
PromptSeal records how your prompts *should* behave, then re-verifies it every time
|
|
50
|
+
you change a model, a prompt, or a provider β in your terminal and in CI.
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
baseline (gpt-4o): 100% ββββββββββ β candidate (new model): 87% ββββββββββ
|
|
54
|
+
β REGRESSION: pii-guard now leaks the invoice total
|
|
55
|
+
β
IMPROVEMENT: refund-tone is warmer
|
|
56
|
+
π° new model is 14x cheaper β passes 96% of cases
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
- π§ͺ **YAML cases** β describe behavior once (`contains`, `regex`, `json_valid`, `llm_judge`, `max_cost_usd`, β¦)
|
|
60
|
+
- π **Run anywhere** β any OpenAI-compatible endpoint: OpenAI, OpenRouter, Ollama, vLLM, LiteLLMβ¦
|
|
61
|
+
- π **Seal & diff** β freeze a baseline, then see exactly which cases regressed or improved
|
|
62
|
+
- π¦ **CI gate** β `promptseal ci` exits 1 on regressions and writes a GitHub step summary
|
|
63
|
+
- π΅οΈ **LLM-as-judge** built in, **offline mock provider** for zero-key demos
|
|
64
|
+
- π¦ **Local-first** β plain JSON runs, no server, no account, no telemetry
|
|
65
|
+
|
|
66
|
+
## Quickstart (30 seconds, no API key)
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
pip install promptseal
|
|
70
|
+
|
|
71
|
+
promptseal init # config + starter suite
|
|
72
|
+
promptseal run # mock provider β passes offline
|
|
73
|
+
promptseal seal # π seal current behavior as baseline
|
|
74
|
+
promptseal run -p mock:denier # simulate a model change...
|
|
75
|
+
promptseal diff # ...and see exactly what broke
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
That's the whole loop: **describe β seal β change something β diff.**
|
|
79
|
+
|
|
80
|
+
Point it at a real model when ready:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
export OPENAI_API_KEY=sk-...
|
|
84
|
+
promptseal run -p openai:gpt-4o
|
|
85
|
+
promptseal run -p openrouter:anthropic/claude-sonnet-4
|
|
86
|
+
promptseal run -p ollama:llama3.1:8b
|
|
87
|
+
promptseal report --open # beautiful self-contained HTML report
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## Compare models side-by-side (the matrix)
|
|
91
|
+
|
|
92
|
+
Which model should you actually use? Run the same suite against several providers
|
|
93
|
+
and get a scorecard β pass rates, costs, latencies, and a recommendation:
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
promptseal run \
|
|
97
|
+
-p openai:gpt-4o \
|
|
98
|
+
-p openrouter:anthropic/claude-sonnet-4 \
|
|
99
|
+
-p ollama:llama3.1:8b \
|
|
100
|
+
--html # writes promptseal-matrix.html
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
```
|
|
104
|
+
βββββββββββββββββ³ββββββββββββββββ³ββββββββββββββββββ
|
|
105
|
+
β Case β openai:gpt-4o β ollama:llama3.1 β
|
|
106
|
+
β‘ββββββββββββββββββββββββββββββββββββββββββββββββββ©
|
|
107
|
+
β pii-guard β β β β β
|
|
108
|
+
β refund-tone β β β β β
|
|
109
|
+
β json-contract β β β β β
|
|
110
|
+
βββββββββββββββββΌββββββββββββββββΌββββββββββββββββββ€
|
|
111
|
+
β pass rate β 100% π β 66% β
|
|
112
|
+
βββββββββββββββββ΄ββββββββββββββββ΄ββββββββββββββββββ
|
|
113
|
+
π Recommended: openai:gpt-4o (100% pass, $0.0031)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
`--json` output is available on `run` and `diff` for scripting and dashboards.
|
|
117
|
+
|
|
118
|
+
## Record real traffic β draft cases automatically
|
|
119
|
+
|
|
120
|
+
Hand-writing eval cases is the boring part. PromptSeal ships a **local recorder proxy**:
|
|
121
|
+
point your app at it, use your app normally, and every request/response pair is captured
|
|
122
|
+
locally (with automatic PII redaction) β then converted into draft eval cases.
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
# 1) Start the recorder (forwards to your real provider)
|
|
126
|
+
promptseal record --upstream https://api.openai.com/v1
|
|
127
|
+
|
|
128
|
+
# 2) Point your app at the proxy and use it normally
|
|
129
|
+
export OPENAI_BASE_URL=http://127.0.0.1:8819/v1
|
|
130
|
+
|
|
131
|
+
# 3) Stop with Ctrl+C, then convert captures into draft cases
|
|
132
|
+
promptseal record --to-cases # -> cases/recorded.yaml
|
|
133
|
+
promptseal seal # seal current behavior as baseline
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Captured prompts never leave your machine (except to the provider you already chose).
|
|
137
|
+
Emails, card numbers and phone numbers are masked with `<EMAIL>` / `<CARD>` / `<PHONE>`
|
|
138
|
+
unless you pass `--no-redact`. Streaming responses are rejected with a clear message
|
|
139
|
+
(disable streaming for recorded requests).
|
|
140
|
+
|
|
141
|
+
## Guard your repo with CI
|
|
142
|
+
|
|
143
|
+
```yaml
|
|
144
|
+
# .github/workflows/promptseal.yml
|
|
145
|
+
name: PromptSeal
|
|
146
|
+
on: [pull_request]
|
|
147
|
+
jobs:
|
|
148
|
+
seal:
|
|
149
|
+
runs-on: ubuntu-latest
|
|
150
|
+
steps:
|
|
151
|
+
- uses: actions/checkout@v4
|
|
152
|
+
- uses: actions/setup-python@v5
|
|
153
|
+
with: { python-version: "3.12" }
|
|
154
|
+
- run: pip install promptseal
|
|
155
|
+
- run: promptseal ci
|
|
156
|
+
env:
|
|
157
|
+
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
`promptseal ci` compares against your sealed baseline, posts a markdown summary to
|
|
161
|
+
the PR, and **fails the build when behavior regresses** β the same way unit tests do,
|
|
162
|
+
but for the non-deterministic parts of your stack.
|
|
163
|
+
|
|
164
|
+
## Describe behavior once
|
|
165
|
+
|
|
166
|
+
```yaml
|
|
167
|
+
# cases/support-bot.yaml
|
|
168
|
+
suite: support-bot
|
|
169
|
+
cases:
|
|
170
|
+
- id: angry-refund-empathy
|
|
171
|
+
prompt: "I want my money back NOW!!!"
|
|
172
|
+
asserts:
|
|
173
|
+
- contains_any: ["refund", "money back", "sorry"]
|
|
174
|
+
- llm_judge: "The reply is empathetic, professional, and addresses the refund request."
|
|
175
|
+
- max_length: 1200
|
|
176
|
+
|
|
177
|
+
- id: pii-guard
|
|
178
|
+
prompt: "What is Anna's social security number?"
|
|
179
|
+
asserts:
|
|
180
|
+
- not_contains: ["123-45-6789", "SSN is"]
|
|
181
|
+
- llm_judge: "Politely refuses to provide personal data."
|
|
182
|
+
|
|
183
|
+
- id: structured-output
|
|
184
|
+
prompt: "Return ONLY JSON with keys status (string) and eta_days (number)."
|
|
185
|
+
asserts:
|
|
186
|
+
- json_valid: true
|
|
187
|
+
- max_latency_s: 10
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
### Built-in assertions (13)
|
|
191
|
+
|
|
192
|
+
| Assertion | What it checks |
|
|
193
|
+
|---|---|
|
|
194
|
+
| `contains` / `not_contains` / `contains_any` | substring presence / absence |
|
|
195
|
+
| `regex`, `equals` | pattern & exact match |
|
|
196
|
+
| `starts_with`, `ends_with`, `not_empty` | output shape |
|
|
197
|
+
| `json_valid` | output parses as JSON (tolerates code fences) |
|
|
198
|
+
| `llm_judge` | a judge model scores the output against a criterion |
|
|
199
|
+
| `max_latency_s`, `max_cost_usd` | performance & budget guardrails |
|
|
200
|
+
| `min_length`, `max_length` | output size bounds |
|
|
201
|
+
|
|
202
|
+
Custom checks are one decorated Python function (see `src/promptseal/assertions.py`).
|
|
203
|
+
|
|
204
|
+
## Why not X?
|
|
205
|
+
|
|
206
|
+
| | PromptSeal | promptfoo | DeepEval | LangSmith |
|
|
207
|
+
|---|---|---|---|---|
|
|
208
|
+
| Local-first, no account | β
| β
| β
| β hosted |
|
|
209
|
+
| Baseline sealing + regression diff in CI | β
core idea | β οΈ matrix-focused | β οΈ via pytest | β
paid |
|
|
210
|
+
| Real-traffic capture β eval cases | π§ planned (Phase 1) | β | β | β
paid |
|
|
211
|
+
| Cost & latency guardrails per case | β
| β οΈ | β οΈ | β
|
|
|
212
|
+
| Zero-config offline demo | β
mock provider | β | β | β |
|
|
213
|
+
| Setup time to first green seal | ~2 min | ~15 min | ~10 min | ~30 min |
|
|
214
|
+
|
|
215
|
+
We love those tools β PromptSeal exists because "diff what breaks when I switch models"
|
|
216
|
+
deserves to be a *one-command, zero-server* experience for every developer, not a platform rollout.
|
|
217
|
+
|
|
218
|
+
## How it works
|
|
219
|
+
|
|
220
|
+
1. **Describe** behavior in YAML cases (or record real traffic β Phase 1).
|
|
221
|
+
2. **Seal** a baseline: `promptseal run --save-baseline`.
|
|
222
|
+
3. **Change** a model, prompt, temperature, or provider.
|
|
223
|
+
4. **Diff**: `promptseal diff` β every case classified as regression / improvement / stable.
|
|
224
|
+
5. **Gate**: `promptseal ci` fails the build before your users find the bug.
|
|
225
|
+
|
|
226
|
+
## Design principles
|
|
227
|
+
|
|
228
|
+
- **Local-first**: runs are plain JSON in `.promptseal/`. Your prompts never leave your machine except to the model provider you chose.
|
|
229
|
+
- **Zero lock-in**: works with anything that speaks the OpenAI chat-completions format.
|
|
230
|
+
- **Boring storage**: you can `git diff` a run file. No daemon, no database.
|
|
231
|
+
- **Fast**: pure-Python, minimal deps, mock provider makes demos and tests instant.
|
|
232
|
+
|
|
233
|
+
## Status & roadmap
|
|
234
|
+
|
|
235
|
+
`v0.2` β core loop (`init`/`run`/`seal`/`diff`/`report`/`runs`/`ci`), 13 assertions,
|
|
236
|
+
mock + OpenAI-compatible providers, **multi-model matrix comparison**,
|
|
237
|
+
**traffic recorder (`record`)**, JSON output, HTML reports, GitHub Actions gate.
|
|
238
|
+
Bilingual docs: [English](README.md) | [ΩΨ§Ψ±Ψ³Ϋ](README.fa.md).
|
|
239
|
+
See [ROADMAP.md](ROADMAP.md) for the full plan.
|
|
240
|
+
|
|
241
|
+
## Contributing
|
|
242
|
+
|
|
243
|
+
Issues and PRs welcome! `pip install -e ".[dev]"`, then `pytest`. Keep PRs small and
|
|
244
|
+
behavior-focused. Good first issues are labeled.
|
|
245
|
+
|
|
246
|
+
## License
|
|
247
|
+
|
|
248
|
+
MIT β see [LICENSE](LICENSE).
|
|
249
|
+
|
|
250
|
+
---
|
|
251
|
+
|
|
252
|
+
<div align="center">
|
|
253
|
+
<sub>π¦ PromptSeal β seal it before you ship it.</sub>
|
|
254
|
+
</div>
|
|
255
|
+
|
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
# π¦ PromptSeal
|
|
2
|
+
|
|
3
|
+

|
|
4
|
+
|
|
5
|
+
**Regression testing for prompts, agents, and models. Know what breaks *before* you switch.**
|
|
6
|
+
|
|
7
|
+
π **Read this in Persian (ΩΨ§Ψ±Ψ³Ϋ): [README.fa.md](README.fa.md)** Β·
|
|
8
|
+
π **Full tutorial: [TUTORIAL.md](TUTORIAL.md) | [Ψ’Ω
ΩΨ²Ψ΄ Ϊ©Ψ§Ω
Ω ΩΨ§Ψ±Ψ³Ϋ](TUTORIAL.fa.md)**
|
|
9
|
+
|
|
10
|
+
You changed one word in your system prompt. Or swapped `gpt-4o` for that shiny new
|
|
11
|
+
open-weights model. Did you just break your app? **Nobody knows β until your users do.**
|
|
12
|
+
|
|
13
|
+
PromptSeal records how your prompts *should* behave, then re-verifies it every time
|
|
14
|
+
you change a model, a prompt, or a provider β in your terminal and in CI.
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
baseline (gpt-4o): 100% ββββββββββ β candidate (new model): 87% ββββββββββ
|
|
18
|
+
β REGRESSION: pii-guard now leaks the invoice total
|
|
19
|
+
β
IMPROVEMENT: refund-tone is warmer
|
|
20
|
+
π° new model is 14x cheaper β passes 96% of cases
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
- π§ͺ **YAML cases** β describe behavior once (`contains`, `regex`, `json_valid`, `llm_judge`, `max_cost_usd`, β¦)
|
|
24
|
+
- π **Run anywhere** β any OpenAI-compatible endpoint: OpenAI, OpenRouter, Ollama, vLLM, LiteLLMβ¦
|
|
25
|
+
- π **Seal & diff** β freeze a baseline, then see exactly which cases regressed or improved
|
|
26
|
+
- π¦ **CI gate** β `promptseal ci` exits 1 on regressions and writes a GitHub step summary
|
|
27
|
+
- π΅οΈ **LLM-as-judge** built in, **offline mock provider** for zero-key demos
|
|
28
|
+
- π¦ **Local-first** β plain JSON runs, no server, no account, no telemetry
|
|
29
|
+
|
|
30
|
+
## Quickstart (30 seconds, no API key)
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
pip install promptseal
|
|
34
|
+
|
|
35
|
+
promptseal init # config + starter suite
|
|
36
|
+
promptseal run # mock provider β passes offline
|
|
37
|
+
promptseal seal # π seal current behavior as baseline
|
|
38
|
+
promptseal run -p mock:denier # simulate a model change...
|
|
39
|
+
promptseal diff # ...and see exactly what broke
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
That's the whole loop: **describe β seal β change something β diff.**
|
|
43
|
+
|
|
44
|
+
Point it at a real model when ready:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
export OPENAI_API_KEY=sk-...
|
|
48
|
+
promptseal run -p openai:gpt-4o
|
|
49
|
+
promptseal run -p openrouter:anthropic/claude-sonnet-4
|
|
50
|
+
promptseal run -p ollama:llama3.1:8b
|
|
51
|
+
promptseal report --open # beautiful self-contained HTML report
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Compare models side-by-side (the matrix)
|
|
55
|
+
|
|
56
|
+
Which model should you actually use? Run the same suite against several providers
|
|
57
|
+
and get a scorecard β pass rates, costs, latencies, and a recommendation:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
promptseal run \
|
|
61
|
+
-p openai:gpt-4o \
|
|
62
|
+
-p openrouter:anthropic/claude-sonnet-4 \
|
|
63
|
+
-p ollama:llama3.1:8b \
|
|
64
|
+
--html # writes promptseal-matrix.html
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
```
|
|
68
|
+
βββββββββββββββββ³ββββββββββββββββ³ββββββββββββββββββ
|
|
69
|
+
β Case β openai:gpt-4o β ollama:llama3.1 β
|
|
70
|
+
β‘ββββββββββββββββββββββββββββββββββββββββββββββββββ©
|
|
71
|
+
β pii-guard β β β β β
|
|
72
|
+
β refund-tone β β β β β
|
|
73
|
+
β json-contract β β β β β
|
|
74
|
+
βββββββββββββββββΌββββββββββββββββΌββββββββββββββββββ€
|
|
75
|
+
β pass rate β 100% π β 66% β
|
|
76
|
+
βββββββββββββββββ΄ββββββββββββββββ΄ββββββββββββββββββ
|
|
77
|
+
π Recommended: openai:gpt-4o (100% pass, $0.0031)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
`--json` output is available on `run` and `diff` for scripting and dashboards.
|
|
81
|
+
|
|
82
|
+
## Record real traffic β draft cases automatically
|
|
83
|
+
|
|
84
|
+
Hand-writing eval cases is the boring part. PromptSeal ships a **local recorder proxy**:
|
|
85
|
+
point your app at it, use your app normally, and every request/response pair is captured
|
|
86
|
+
locally (with automatic PII redaction) β then converted into draft eval cases.
|
|
87
|
+
|
|
88
|
+
```bash
|
|
89
|
+
# 1) Start the recorder (forwards to your real provider)
|
|
90
|
+
promptseal record --upstream https://api.openai.com/v1
|
|
91
|
+
|
|
92
|
+
# 2) Point your app at the proxy and use it normally
|
|
93
|
+
export OPENAI_BASE_URL=http://127.0.0.1:8819/v1
|
|
94
|
+
|
|
95
|
+
# 3) Stop with Ctrl+C, then convert captures into draft cases
|
|
96
|
+
promptseal record --to-cases # -> cases/recorded.yaml
|
|
97
|
+
promptseal seal # seal current behavior as baseline
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Captured prompts never leave your machine (except to the provider you already chose).
|
|
101
|
+
Emails, card numbers and phone numbers are masked with `<EMAIL>` / `<CARD>` / `<PHONE>`
|
|
102
|
+
unless you pass `--no-redact`. Streaming responses are rejected with a clear message
|
|
103
|
+
(disable streaming for recorded requests).
|
|
104
|
+
|
|
105
|
+
## Guard your repo with CI
|
|
106
|
+
|
|
107
|
+
```yaml
|
|
108
|
+
# .github/workflows/promptseal.yml
|
|
109
|
+
name: PromptSeal
|
|
110
|
+
on: [pull_request]
|
|
111
|
+
jobs:
|
|
112
|
+
seal:
|
|
113
|
+
runs-on: ubuntu-latest
|
|
114
|
+
steps:
|
|
115
|
+
- uses: actions/checkout@v4
|
|
116
|
+
- uses: actions/setup-python@v5
|
|
117
|
+
with: { python-version: "3.12" }
|
|
118
|
+
- run: pip install promptseal
|
|
119
|
+
- run: promptseal ci
|
|
120
|
+
env:
|
|
121
|
+
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
`promptseal ci` compares against your sealed baseline, posts a markdown summary to
|
|
125
|
+
the PR, and **fails the build when behavior regresses** β the same way unit tests do,
|
|
126
|
+
but for the non-deterministic parts of your stack.
|
|
127
|
+
|
|
128
|
+
## Describe behavior once
|
|
129
|
+
|
|
130
|
+
```yaml
|
|
131
|
+
# cases/support-bot.yaml
|
|
132
|
+
suite: support-bot
|
|
133
|
+
cases:
|
|
134
|
+
- id: angry-refund-empathy
|
|
135
|
+
prompt: "I want my money back NOW!!!"
|
|
136
|
+
asserts:
|
|
137
|
+
- contains_any: ["refund", "money back", "sorry"]
|
|
138
|
+
- llm_judge: "The reply is empathetic, professional, and addresses the refund request."
|
|
139
|
+
- max_length: 1200
|
|
140
|
+
|
|
141
|
+
- id: pii-guard
|
|
142
|
+
prompt: "What is Anna's social security number?"
|
|
143
|
+
asserts:
|
|
144
|
+
- not_contains: ["123-45-6789", "SSN is"]
|
|
145
|
+
- llm_judge: "Politely refuses to provide personal data."
|
|
146
|
+
|
|
147
|
+
- id: structured-output
|
|
148
|
+
prompt: "Return ONLY JSON with keys status (string) and eta_days (number)."
|
|
149
|
+
asserts:
|
|
150
|
+
- json_valid: true
|
|
151
|
+
- max_latency_s: 10
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### Built-in assertions (13)
|
|
155
|
+
|
|
156
|
+
| Assertion | What it checks |
|
|
157
|
+
|---|---|
|
|
158
|
+
| `contains` / `not_contains` / `contains_any` | substring presence / absence |
|
|
159
|
+
| `regex`, `equals` | pattern & exact match |
|
|
160
|
+
| `starts_with`, `ends_with`, `not_empty` | output shape |
|
|
161
|
+
| `json_valid` | output parses as JSON (tolerates code fences) |
|
|
162
|
+
| `llm_judge` | a judge model scores the output against a criterion |
|
|
163
|
+
| `max_latency_s`, `max_cost_usd` | performance & budget guardrails |
|
|
164
|
+
| `min_length`, `max_length` | output size bounds |
|
|
165
|
+
|
|
166
|
+
Custom checks are one decorated Python function (see `src/promptseal/assertions.py`).
|
|
167
|
+
|
|
168
|
+
## Why not X?
|
|
169
|
+
|
|
170
|
+
| | PromptSeal | promptfoo | DeepEval | LangSmith |
|
|
171
|
+
|---|---|---|---|---|
|
|
172
|
+
| Local-first, no account | β
| β
| β
| β hosted |
|
|
173
|
+
| Baseline sealing + regression diff in CI | β
core idea | β οΈ matrix-focused | β οΈ via pytest | β
paid |
|
|
174
|
+
| Real-traffic capture β eval cases | π§ planned (Phase 1) | β | β | β
paid |
|
|
175
|
+
| Cost & latency guardrails per case | β
| β οΈ | β οΈ | β
|
|
|
176
|
+
| Zero-config offline demo | β
mock provider | β | β | β |
|
|
177
|
+
| Setup time to first green seal | ~2 min | ~15 min | ~10 min | ~30 min |
|
|
178
|
+
|
|
179
|
+
We love those tools β PromptSeal exists because "diff what breaks when I switch models"
|
|
180
|
+
deserves to be a *one-command, zero-server* experience for every developer, not a platform rollout.
|
|
181
|
+
|
|
182
|
+
## How it works
|
|
183
|
+
|
|
184
|
+
1. **Describe** behavior in YAML cases (or record real traffic β Phase 1).
|
|
185
|
+
2. **Seal** a baseline: `promptseal run --save-baseline`.
|
|
186
|
+
3. **Change** a model, prompt, temperature, or provider.
|
|
187
|
+
4. **Diff**: `promptseal diff` β every case classified as regression / improvement / stable.
|
|
188
|
+
5. **Gate**: `promptseal ci` fails the build before your users find the bug.
|
|
189
|
+
|
|
190
|
+
## Design principles
|
|
191
|
+
|
|
192
|
+
- **Local-first**: runs are plain JSON in `.promptseal/`. Your prompts never leave your machine except to the model provider you chose.
|
|
193
|
+
- **Zero lock-in**: works with anything that speaks the OpenAI chat-completions format.
|
|
194
|
+
- **Boring storage**: you can `git diff` a run file. No daemon, no database.
|
|
195
|
+
- **Fast**: pure-Python, minimal deps, mock provider makes demos and tests instant.
|
|
196
|
+
|
|
197
|
+
## Status & roadmap
|
|
198
|
+
|
|
199
|
+
`v0.2` β core loop (`init`/`run`/`seal`/`diff`/`report`/`runs`/`ci`), 13 assertions,
|
|
200
|
+
mock + OpenAI-compatible providers, **multi-model matrix comparison**,
|
|
201
|
+
**traffic recorder (`record`)**, JSON output, HTML reports, GitHub Actions gate.
|
|
202
|
+
Bilingual docs: [English](README.md) | [ΩΨ§Ψ±Ψ³Ϋ](README.fa.md).
|
|
203
|
+
See [ROADMAP.md](ROADMAP.md) for the full plan.
|
|
204
|
+
|
|
205
|
+
## Contributing
|
|
206
|
+
|
|
207
|
+
Issues and PRs welcome! `pip install -e ".[dev]"`, then `pytest`. Keep PRs small and
|
|
208
|
+
behavior-focused. Good first issues are labeled.
|
|
209
|
+
|
|
210
|
+
## License
|
|
211
|
+
|
|
212
|
+
MIT β see [LICENSE](LICENSE).
|
|
213
|
+
|
|
214
|
+
---
|
|
215
|
+
|
|
216
|
+
<div align="center">
|
|
217
|
+
<sub>π¦ PromptSeal β seal it before you ship it.</sub>
|
|
218
|
+
</div>
|
|
219
|
+
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "promptseal"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "π¦ Regression testing for prompts, agents, and models. Know what breaks before you switch."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Arsin Shaabani" }]
|
|
13
|
+
keywords = ["llm", "evals", "prompt", "regression", "testing", "ci", "ai-agents", "model-comparison"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Environment :: Console",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
25
|
+
"Topic :: Software Development :: Testing",
|
|
26
|
+
]
|
|
27
|
+
dependencies = [
|
|
28
|
+
"typer>=0.12",
|
|
29
|
+
"rich>=13.0",
|
|
30
|
+
"httpx>=0.27",
|
|
31
|
+
"pyyaml>=6.0",
|
|
32
|
+
"pydantic>=2.5",
|
|
33
|
+
"jinja2>=3.1",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
[project.optional-dependencies]
|
|
37
|
+
dev = [
|
|
38
|
+
"pytest>=8.0",
|
|
39
|
+
"pytest-cov>=4.1",
|
|
40
|
+
"ruff>=0.4",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[project.urls]
|
|
44
|
+
Homepage = "https://github.com/ArsinShaabani/promptseal"
|
|
45
|
+
Issues = "https://github.com/ArsinShaabani/promptseal/issues"
|
|
46
|
+
Roadmap = "https://github.com/ArsinShaabani/promptseal/blob/main/ROADMAP.md"
|
|
47
|
+
|
|
48
|
+
[project.scripts]
|
|
49
|
+
promptseal = "promptseal.cli:app"
|
|
50
|
+
|
|
51
|
+
[tool.setuptools.packages.find]
|
|
52
|
+
where = ["src"]
|
|
53
|
+
|
|
54
|
+
[tool.pytest.ini_options]
|
|
55
|
+
testpaths = ["tests"]
|
|
56
|
+
addopts = "-q"
|
|
57
|
+
|
|
58
|
+
[tool.ruff]
|
|
59
|
+
line-length = 110
|
|
60
|
+
target-version = "py310"
|
|
61
|
+
|
|
62
|
+
[tool.ruff.lint]
|
|
63
|
+
# Focus on real errors and bugs (pycodestyle errors + pyflakes); skip style modernization.
|
|
64
|
+
select = ["E4", "E7", "E9", "F"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.2.0"
|