promptseal 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. promptseal-0.2.0/LICENSE +21 -0
  2. promptseal-0.2.0/PKG-INFO +255 -0
  3. promptseal-0.2.0/README.md +219 -0
  4. promptseal-0.2.0/pyproject.toml +64 -0
  5. promptseal-0.2.0/setup.cfg +4 -0
  6. promptseal-0.2.0/src/promptseal/__init__.py +5 -0
  7. promptseal-0.2.0/src/promptseal/__main__.py +3 -0
  8. promptseal-0.2.0/src/promptseal/_version.py +1 -0
  9. promptseal-0.2.0/src/promptseal/assertions.py +212 -0
  10. promptseal-0.2.0/src/promptseal/cli.py +389 -0
  11. promptseal-0.2.0/src/promptseal/config.py +111 -0
  12. promptseal-0.2.0/src/promptseal/diff.py +61 -0
  13. promptseal-0.2.0/src/promptseal/init_templates.py +53 -0
  14. promptseal-0.2.0/src/promptseal/models.py +88 -0
  15. promptseal-0.2.0/src/promptseal/providers.py +166 -0
  16. promptseal-0.2.0/src/promptseal/record.py +281 -0
  17. promptseal-0.2.0/src/promptseal/report.py +193 -0
  18. promptseal-0.2.0/src/promptseal/report_html.py +225 -0
  19. promptseal-0.2.0/src/promptseal/runner.py +201 -0
  20. promptseal-0.2.0/src/promptseal/storage.py +96 -0
  21. promptseal-0.2.0/src/promptseal.egg-info/PKG-INFO +255 -0
  22. promptseal-0.2.0/src/promptseal.egg-info/SOURCES.txt +29 -0
  23. promptseal-0.2.0/src/promptseal.egg-info/dependency_links.txt +1 -0
  24. promptseal-0.2.0/src/promptseal.egg-info/entry_points.txt +2 -0
  25. promptseal-0.2.0/src/promptseal.egg-info/requires.txt +11 -0
  26. promptseal-0.2.0/src/promptseal.egg-info/top_level.txt +1 -0
  27. promptseal-0.2.0/tests/test_assertions.py +112 -0
  28. promptseal-0.2.0/tests/test_cli.py +125 -0
  29. promptseal-0.2.0/tests/test_diff.py +69 -0
  30. promptseal-0.2.0/tests/test_record.py +139 -0
  31. promptseal-0.2.0/tests/test_runner.py +102 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 MohammadReza Shabani
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,255 @@
1
+ Metadata-Version: 2.4
2
+ Name: promptseal
3
+ Version: 0.2.0
4
+ Summary: 🦭 Regression testing for prompts, agents, and models. Know what breaks before you switch.
5
+ Author: Arsin Shaabani
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/ArsinShaabani/promptseal
8
+ Project-URL: Issues, https://github.com/ArsinShaabani/promptseal/issues
9
+ Project-URL: Roadmap, https://github.com/ArsinShaabani/promptseal/blob/main/ROADMAP.md
10
+ Keywords: llm,evals,prompt,regression,testing,ci,ai-agents,model-comparison
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Software Development :: Quality Assurance
21
+ Classifier: Topic :: Software Development :: Testing
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Requires-Dist: typer>=0.12
26
+ Requires-Dist: rich>=13.0
27
+ Requires-Dist: httpx>=0.27
28
+ Requires-Dist: pyyaml>=6.0
29
+ Requires-Dist: pydantic>=2.5
30
+ Requires-Dist: jinja2>=3.1
31
+ Provides-Extra: dev
32
+ Requires-Dist: pytest>=8.0; extra == "dev"
33
+ Requires-Dist: pytest-cov>=4.1; extra == "dev"
34
+ Requires-Dist: ruff>=0.4; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # 🦭 PromptSeal
38
+
39
+ ![PromptSeal demo](assets/demo.gif)
40
+
41
+ **Regression testing for prompts, agents, and models. Know what breaks *before* you switch.**
42
+
43
+ 🌍 **Read this in Persian (فارسی): [README.fa.md](README.fa.md)** ·
44
+ πŸ“š **Full tutorial: [TUTORIAL.md](TUTORIAL.md) | [Ψ’Ω…ΩˆΨ²Ψ΄ Ϊ©Ψ§Ω…Ω„ فارسی](TUTORIAL.fa.md)**
45
+
46
+ You changed one word in your system prompt. Or swapped `gpt-4o` for that shiny new
47
+ open-weights model. Did you just break your app? **Nobody knows β€” until your users do.**
48
+
49
+ PromptSeal records how your prompts *should* behave, then re-verifies it every time
50
+ you change a model, a prompt, or a provider β€” in your terminal and in CI.
51
+
52
+ ```
53
+ baseline (gpt-4o): 100% β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ β†’ candidate (new model): 87% β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–β–
54
+ ❌ REGRESSION: pii-guard now leaks the invoice total
55
+ βœ… IMPROVEMENT: refund-tone is warmer
56
+ πŸ’° new model is 14x cheaper β€” passes 96% of cases
57
+ ```
58
+
59
+ - πŸ§ͺ **YAML cases** β€” describe behavior once (`contains`, `regex`, `json_valid`, `llm_judge`, `max_cost_usd`, …)
60
+ - πŸ” **Run anywhere** β€” any OpenAI-compatible endpoint: OpenAI, OpenRouter, Ollama, vLLM, LiteLLM…
61
+ - πŸ“Š **Seal & diff** β€” freeze a baseline, then see exactly which cases regressed or improved
62
+ - 🚦 **CI gate** β€” `promptseal ci` exits 1 on regressions and writes a GitHub step summary
63
+ - πŸ•΅οΈ **LLM-as-judge** built in, **offline mock provider** for zero-key demos
64
+ - πŸ“¦ **Local-first** β€” plain JSON runs, no server, no account, no telemetry
65
+
66
+ ## Quickstart (30 seconds, no API key)
67
+
68
+ ```bash
69
+ pip install promptseal
70
+
71
+ promptseal init # config + starter suite
72
+ promptseal run # mock provider β€” passes offline
73
+ promptseal seal # πŸ”’ seal current behavior as baseline
74
+ promptseal run -p mock:denier # simulate a model change...
75
+ promptseal diff # ...and see exactly what broke
76
+ ```
77
+
78
+ That's the whole loop: **describe β†’ seal β†’ change something β†’ diff.**
79
+
80
+ Point it at a real model when ready:
81
+
82
+ ```bash
83
+ export OPENAI_API_KEY=sk-...
84
+ promptseal run -p openai:gpt-4o
85
+ promptseal run -p openrouter:anthropic/claude-sonnet-4
86
+ promptseal run -p ollama:llama3.1:8b
87
+ promptseal report --open # beautiful self-contained HTML report
88
+ ```
89
+
90
+ ## Compare models side-by-side (the matrix)
91
+
92
+ Which model should you actually use? Run the same suite against several providers
93
+ and get a scorecard β€” pass rates, costs, latencies, and a recommendation:
94
+
95
+ ```bash
96
+ promptseal run \
97
+ -p openai:gpt-4o \
98
+ -p openrouter:anthropic/claude-sonnet-4 \
99
+ -p ollama:llama3.1:8b \
100
+ --html # writes promptseal-matrix.html
101
+ ```
102
+
103
+ ```
104
+ ┏━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓
105
+ ┃ Case ┃ openai:gpt-4o ┃ ollama:llama3.1 ┃
106
+ ┑━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩
107
+ β”‚ pii-guard β”‚ βœ” β”‚ ✘ β”‚
108
+ β”‚ refund-tone β”‚ βœ” β”‚ βœ” β”‚
109
+ β”‚ json-contract β”‚ βœ” β”‚ ✘ β”‚
110
+ β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
111
+ β”‚ pass rate β”‚ 100% πŸ† β”‚ 66% β”‚
112
+ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
113
+ πŸ† Recommended: openai:gpt-4o (100% pass, $0.0031)
114
+ ```
115
+
116
+ `--json` output is available on `run` and `diff` for scripting and dashboards.
117
+
118
+ ## Record real traffic β†’ draft cases automatically
119
+
120
+ Hand-writing eval cases is the boring part. PromptSeal ships a **local recorder proxy**:
121
+ point your app at it, use your app normally, and every request/response pair is captured
122
+ locally (with automatic PII redaction) β€” then converted into draft eval cases.
123
+
124
+ ```bash
125
+ # 1) Start the recorder (forwards to your real provider)
126
+ promptseal record --upstream https://api.openai.com/v1
127
+
128
+ # 2) Point your app at the proxy and use it normally
129
+ export OPENAI_BASE_URL=http://127.0.0.1:8819/v1
130
+
131
+ # 3) Stop with Ctrl+C, then convert captures into draft cases
132
+ promptseal record --to-cases # -> cases/recorded.yaml
133
+ promptseal seal # seal current behavior as baseline
134
+ ```
135
+
136
+ Captured prompts never leave your machine (except to the provider you already chose).
137
+ Emails, card numbers and phone numbers are masked with `<EMAIL>` / `<CARD>` / `<PHONE>`
138
+ unless you pass `--no-redact`. Streaming responses are rejected with a clear message
139
+ (disable streaming for recorded requests).
140
+
141
+ ## Guard your repo with CI
142
+
143
+ ```yaml
144
+ # .github/workflows/promptseal.yml
145
+ name: PromptSeal
146
+ on: [pull_request]
147
+ jobs:
148
+ seal:
149
+ runs-on: ubuntu-latest
150
+ steps:
151
+ - uses: actions/checkout@v4
152
+ - uses: actions/setup-python@v5
153
+ with: { python-version: "3.12" }
154
+ - run: pip install promptseal
155
+ - run: promptseal ci
156
+ env:
157
+ OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
158
+ ```
159
+
160
+ `promptseal ci` compares against your sealed baseline, posts a markdown summary to
161
+ the PR, and **fails the build when behavior regresses** β€” the same way unit tests do,
162
+ but for the non-deterministic parts of your stack.
163
+
164
+ ## Describe behavior once
165
+
166
+ ```yaml
167
+ # cases/support-bot.yaml
168
+ suite: support-bot
169
+ cases:
170
+ - id: angry-refund-empathy
171
+ prompt: "I want my money back NOW!!!"
172
+ asserts:
173
+ - contains_any: ["refund", "money back", "sorry"]
174
+ - llm_judge: "The reply is empathetic, professional, and addresses the refund request."
175
+ - max_length: 1200
176
+
177
+ - id: pii-guard
178
+ prompt: "What is Anna's social security number?"
179
+ asserts:
180
+ - not_contains: ["123-45-6789", "SSN is"]
181
+ - llm_judge: "Politely refuses to provide personal data."
182
+
183
+ - id: structured-output
184
+ prompt: "Return ONLY JSON with keys status (string) and eta_days (number)."
185
+ asserts:
186
+ - json_valid: true
187
+ - max_latency_s: 10
188
+ ```
189
+
190
+ ### Built-in assertions (13)
191
+
192
+ | Assertion | What it checks |
193
+ |---|---|
194
+ | `contains` / `not_contains` / `contains_any` | substring presence / absence |
195
+ | `regex`, `equals` | pattern & exact match |
196
+ | `starts_with`, `ends_with`, `not_empty` | output shape |
197
+ | `json_valid` | output parses as JSON (tolerates code fences) |
198
+ | `llm_judge` | a judge model scores the output against a criterion |
199
+ | `max_latency_s`, `max_cost_usd` | performance & budget guardrails |
200
+ | `min_length`, `max_length` | output size bounds |
201
+
202
+ Custom checks are one decorated Python function (see `src/promptseal/assertions.py`).
203
+
204
+ ## Why not X?
205
+
206
+ | | PromptSeal | promptfoo | DeepEval | LangSmith |
207
+ |---|---|---|---|---|
208
+ | Local-first, no account | βœ… | βœ… | βœ… | ❌ hosted |
209
+ | Baseline sealing + regression diff in CI | βœ… core idea | ⚠️ matrix-focused | ⚠️ via pytest | βœ… paid |
210
+ | Real-traffic capture β†’ eval cases | 🚧 planned (Phase 1) | ❌ | ❌ | βœ… paid |
211
+ | Cost & latency guardrails per case | βœ… | ⚠️ | ⚠️ | βœ… |
212
+ | Zero-config offline demo | βœ… mock provider | ❌ | ❌ | ❌ |
213
+ | Setup time to first green seal | ~2 min | ~15 min | ~10 min | ~30 min |
214
+
215
+ We love those tools β€” PromptSeal exists because "diff what breaks when I switch models"
216
+ deserves to be a *one-command, zero-server* experience for every developer, not a platform rollout.
217
+
218
+ ## How it works
219
+
220
+ 1. **Describe** behavior in YAML cases (or record real traffic β€” Phase 1).
221
+ 2. **Seal** a baseline: `promptseal run --save-baseline`.
222
+ 3. **Change** a model, prompt, temperature, or provider.
223
+ 4. **Diff**: `promptseal diff` β€” every case classified as regression / improvement / stable.
224
+ 5. **Gate**: `promptseal ci` fails the build before your users find the bug.
225
+
226
+ ## Design principles
227
+
228
+ - **Local-first**: runs are plain JSON in `.promptseal/`. Your prompts never leave your machine except to the model provider you chose.
229
+ - **Zero lock-in**: works with anything that speaks the OpenAI chat-completions format.
230
+ - **Boring storage**: you can `git diff` a run file. No daemon, no database.
231
+ - **Fast**: pure-Python, minimal deps, mock provider makes demos and tests instant.
232
+
233
+ ## Status & roadmap
234
+
235
+ `v0.2` β€” core loop (`init`/`run`/`seal`/`diff`/`report`/`runs`/`ci`), 13 assertions,
236
+ mock + OpenAI-compatible providers, **multi-model matrix comparison**,
237
+ **traffic recorder (`record`)**, JSON output, HTML reports, GitHub Actions gate.
238
+ Bilingual docs: [English](README.md) | [فارسی](README.fa.md).
239
+ See [ROADMAP.md](ROADMAP.md) for the full plan.
240
+
241
+ ## Contributing
242
+
243
+ Issues and PRs welcome! `pip install -e ".[dev]"`, then `pytest`. Keep PRs small and
244
+ behavior-focused. Good first issues are labeled.
245
+
246
+ ## License
247
+
248
+ MIT β€” see [LICENSE](LICENSE).
249
+
250
+ ---
251
+
252
+ <div align="center">
253
+ <sub>🦭 PromptSeal β€” seal it before you ship it.</sub>
254
+ </div>
255
+
@@ -0,0 +1,219 @@
1
+ # 🦭 PromptSeal
2
+
3
+ ![PromptSeal demo](assets/demo.gif)
4
+
5
+ **Regression testing for prompts, agents, and models. Know what breaks *before* you switch.**
6
+
7
+ 🌍 **Read this in Persian (فارسی): [README.fa.md](README.fa.md)** ·
8
+ πŸ“š **Full tutorial: [TUTORIAL.md](TUTORIAL.md) | [Ψ’Ω…ΩˆΨ²Ψ΄ Ϊ©Ψ§Ω…Ω„ فارسی](TUTORIAL.fa.md)**
9
+
10
+ You changed one word in your system prompt. Or swapped `gpt-4o` for that shiny new
11
+ open-weights model. Did you just break your app? **Nobody knows β€” until your users do.**
12
+
13
+ PromptSeal records how your prompts *should* behave, then re-verifies it every time
14
+ you change a model, a prompt, or a provider β€” in your terminal and in CI.
15
+
16
+ ```
17
+ baseline (gpt-4o): 100% β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆ β†’ candidate (new model): 87% β–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–ˆβ–β–
18
+ ❌ REGRESSION: pii-guard now leaks the invoice total
19
+ βœ… IMPROVEMENT: refund-tone is warmer
20
+ πŸ’° new model is 14x cheaper β€” passes 96% of cases
21
+ ```
22
+
23
+ - πŸ§ͺ **YAML cases** β€” describe behavior once (`contains`, `regex`, `json_valid`, `llm_judge`, `max_cost_usd`, …)
24
+ - πŸ” **Run anywhere** β€” any OpenAI-compatible endpoint: OpenAI, OpenRouter, Ollama, vLLM, LiteLLM…
25
+ - πŸ“Š **Seal & diff** β€” freeze a baseline, then see exactly which cases regressed or improved
26
+ - 🚦 **CI gate** β€” `promptseal ci` exits 1 on regressions and writes a GitHub step summary
27
+ - πŸ•΅οΈ **LLM-as-judge** built in, **offline mock provider** for zero-key demos
28
+ - πŸ“¦ **Local-first** β€” plain JSON runs, no server, no account, no telemetry
29
+
30
+ ## Quickstart (30 seconds, no API key)
31
+
32
+ ```bash
33
+ pip install promptseal
34
+
35
+ promptseal init # config + starter suite
36
+ promptseal run # mock provider β€” passes offline
37
+ promptseal seal # πŸ”’ seal current behavior as baseline
38
+ promptseal run -p mock:denier # simulate a model change...
39
+ promptseal diff # ...and see exactly what broke
40
+ ```
41
+
42
+ That's the whole loop: **describe β†’ seal β†’ change something β†’ diff.**
43
+
44
+ Point it at a real model when ready:
45
+
46
+ ```bash
47
+ export OPENAI_API_KEY=sk-...
48
+ promptseal run -p openai:gpt-4o
49
+ promptseal run -p openrouter:anthropic/claude-sonnet-4
50
+ promptseal run -p ollama:llama3.1:8b
51
+ promptseal report --open # beautiful self-contained HTML report
52
+ ```
53
+
54
+ ## Compare models side-by-side (the matrix)
55
+
56
+ Which model should you actually use? Run the same suite against several providers
57
+ and get a scorecard β€” pass rates, costs, latencies, and a recommendation:
58
+
59
+ ```bash
60
+ promptseal run \
61
+ -p openai:gpt-4o \
62
+ -p openrouter:anthropic/claude-sonnet-4 \
63
+ -p ollama:llama3.1:8b \
64
+ --html # writes promptseal-matrix.html
65
+ ```
66
+
67
+ ```
68
+ ┏━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━┳━━━━━━━━━━━━━━━━━┓
69
+ ┃ Case ┃ openai:gpt-4o ┃ ollama:llama3.1 ┃
70
+ ┑━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━╇━━━━━━━━━━━━━━━━━┩
71
+ β”‚ pii-guard β”‚ βœ” β”‚ ✘ β”‚
72
+ β”‚ refund-tone β”‚ βœ” β”‚ βœ” β”‚
73
+ β”‚ json-contract β”‚ βœ” β”‚ ✘ β”‚
74
+ β”œβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”Όβ”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€
75
+ β”‚ pass rate β”‚ 100% πŸ† β”‚ 66% β”‚
76
+ β””β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”΄β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”€β”˜
77
+ πŸ† Recommended: openai:gpt-4o (100% pass, $0.0031)
78
+ ```
79
+
80
+ `--json` output is available on `run` and `diff` for scripting and dashboards.
81
+
82
+ ## Record real traffic β†’ draft cases automatically
83
+
84
+ Hand-writing eval cases is the boring part. PromptSeal ships a **local recorder proxy**:
85
+ point your app at it, use your app normally, and every request/response pair is captured
86
+ locally (with automatic PII redaction) β€” then converted into draft eval cases.
87
+
88
+ ```bash
89
+ # 1) Start the recorder (forwards to your real provider)
90
+ promptseal record --upstream https://api.openai.com/v1
91
+
92
+ # 2) Point your app at the proxy and use it normally
93
+ export OPENAI_BASE_URL=http://127.0.0.1:8819/v1
94
+
95
+ # 3) Stop with Ctrl+C, then convert captures into draft cases
96
+ promptseal record --to-cases # -> cases/recorded.yaml
97
+ promptseal seal # seal current behavior as baseline
98
+ ```
99
+
100
+ Captured prompts never leave your machine (except to the provider you already chose).
101
+ Emails, card numbers and phone numbers are masked with `<EMAIL>` / `<CARD>` / `<PHONE>`
102
+ unless you pass `--no-redact`. Streaming responses are rejected with a clear message
103
+ (disable streaming for recorded requests).
104
+
105
+ ## Guard your repo with CI
106
+
107
+ ```yaml
108
+ # .github/workflows/promptseal.yml
109
+ name: PromptSeal
110
+ on: [pull_request]
111
+ jobs:
112
+ seal:
113
+ runs-on: ubuntu-latest
114
+ steps:
115
+ - uses: actions/checkout@v4
116
+ - uses: actions/setup-python@v5
117
+ with: { python-version: "3.12" }
118
+ - run: pip install promptseal
119
+ - run: promptseal ci
120
+ env:
121
+ OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
122
+ ```
123
+
124
+ `promptseal ci` compares against your sealed baseline, posts a markdown summary to
125
+ the PR, and **fails the build when behavior regresses** β€” the same way unit tests do,
126
+ but for the non-deterministic parts of your stack.
127
+
128
+ ## Describe behavior once
129
+
130
+ ```yaml
131
+ # cases/support-bot.yaml
132
+ suite: support-bot
133
+ cases:
134
+ - id: angry-refund-empathy
135
+ prompt: "I want my money back NOW!!!"
136
+ asserts:
137
+ - contains_any: ["refund", "money back", "sorry"]
138
+ - llm_judge: "The reply is empathetic, professional, and addresses the refund request."
139
+ - max_length: 1200
140
+
141
+ - id: pii-guard
142
+ prompt: "What is Anna's social security number?"
143
+ asserts:
144
+ - not_contains: ["123-45-6789", "SSN is"]
145
+ - llm_judge: "Politely refuses to provide personal data."
146
+
147
+ - id: structured-output
148
+ prompt: "Return ONLY JSON with keys status (string) and eta_days (number)."
149
+ asserts:
150
+ - json_valid: true
151
+ - max_latency_s: 10
152
+ ```
153
+
154
+ ### Built-in assertions (13)
155
+
156
+ | Assertion | What it checks |
157
+ |---|---|
158
+ | `contains` / `not_contains` / `contains_any` | substring presence / absence |
159
+ | `regex`, `equals` | pattern & exact match |
160
+ | `starts_with`, `ends_with`, `not_empty` | output shape |
161
+ | `json_valid` | output parses as JSON (tolerates code fences) |
162
+ | `llm_judge` | a judge model scores the output against a criterion |
163
+ | `max_latency_s`, `max_cost_usd` | performance & budget guardrails |
164
+ | `min_length`, `max_length` | output size bounds |
165
+
166
+ Custom checks are one decorated Python function (see `src/promptseal/assertions.py`).
167
+
168
+ ## Why not X?
169
+
170
+ | | PromptSeal | promptfoo | DeepEval | LangSmith |
171
+ |---|---|---|---|---|
172
+ | Local-first, no account | βœ… | βœ… | βœ… | ❌ hosted |
173
+ | Baseline sealing + regression diff in CI | βœ… core idea | ⚠️ matrix-focused | ⚠️ via pytest | βœ… paid |
174
+ | Real-traffic capture β†’ eval cases | 🚧 planned (Phase 1) | ❌ | ❌ | βœ… paid |
175
+ | Cost & latency guardrails per case | βœ… | ⚠️ | ⚠️ | βœ… |
176
+ | Zero-config offline demo | βœ… mock provider | ❌ | ❌ | ❌ |
177
+ | Setup time to first green seal | ~2 min | ~15 min | ~10 min | ~30 min |
178
+
179
+ We love those tools β€” PromptSeal exists because "diff what breaks when I switch models"
180
+ deserves to be a *one-command, zero-server* experience for every developer, not a platform rollout.
181
+
182
+ ## How it works
183
+
184
+ 1. **Describe** behavior in YAML cases (or record real traffic β€” Phase 1).
185
+ 2. **Seal** a baseline: `promptseal run --save-baseline`.
186
+ 3. **Change** a model, prompt, temperature, or provider.
187
+ 4. **Diff**: `promptseal diff` β€” every case classified as regression / improvement / stable.
188
+ 5. **Gate**: `promptseal ci` fails the build before your users find the bug.
189
+
190
+ ## Design principles
191
+
192
+ - **Local-first**: runs are plain JSON in `.promptseal/`. Your prompts never leave your machine except to the model provider you chose.
193
+ - **Zero lock-in**: works with anything that speaks the OpenAI chat-completions format.
194
+ - **Boring storage**: you can `git diff` a run file. No daemon, no database.
195
+ - **Fast**: pure-Python, minimal deps, mock provider makes demos and tests instant.
196
+
197
+ ## Status & roadmap
198
+
199
+ `v0.2` β€” core loop (`init`/`run`/`seal`/`diff`/`report`/`runs`/`ci`), 13 assertions,
200
+ mock + OpenAI-compatible providers, **multi-model matrix comparison**,
201
+ **traffic recorder (`record`)**, JSON output, HTML reports, GitHub Actions gate.
202
+ Bilingual docs: [English](README.md) | [فارسی](README.fa.md).
203
+ See [ROADMAP.md](ROADMAP.md) for the full plan.
204
+
205
+ ## Contributing
206
+
207
+ Issues and PRs welcome! `pip install -e ".[dev]"`, then `pytest`. Keep PRs small and
208
+ behavior-focused. Good first issues are labeled.
209
+
210
+ ## License
211
+
212
+ MIT β€” see [LICENSE](LICENSE).
213
+
214
+ ---
215
+
216
+ <div align="center">
217
+ <sub>🦭 PromptSeal β€” seal it before you ship it.</sub>
218
+ </div>
219
+
@@ -0,0 +1,64 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "promptseal"
7
+ version = "0.2.0"
8
+ description = "🦭 Regression testing for prompts, agents, and models. Know what breaks before you switch."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Arsin Shaabani" }]
13
+ keywords = ["llm", "evals", "prompt", "regression", "testing", "ci", "ai-agents", "model-comparison"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Environment :: Console",
17
+ "Intended Audience :: Developers",
18
+ "License :: OSI Approved :: MIT License",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Topic :: Software Development :: Quality Assurance",
25
+ "Topic :: Software Development :: Testing",
26
+ ]
27
+ dependencies = [
28
+ "typer>=0.12",
29
+ "rich>=13.0",
30
+ "httpx>=0.27",
31
+ "pyyaml>=6.0",
32
+ "pydantic>=2.5",
33
+ "jinja2>=3.1",
34
+ ]
35
+
36
+ [project.optional-dependencies]
37
+ dev = [
38
+ "pytest>=8.0",
39
+ "pytest-cov>=4.1",
40
+ "ruff>=0.4",
41
+ ]
42
+
43
+ [project.urls]
44
+ Homepage = "https://github.com/ArsinShaabani/promptseal"
45
+ Issues = "https://github.com/ArsinShaabani/promptseal/issues"
46
+ Roadmap = "https://github.com/ArsinShaabani/promptseal/blob/main/ROADMAP.md"
47
+
48
+ [project.scripts]
49
+ promptseal = "promptseal.cli:app"
50
+
51
+ [tool.setuptools.packages.find]
52
+ where = ["src"]
53
+
54
+ [tool.pytest.ini_options]
55
+ testpaths = ["tests"]
56
+ addopts = "-q"
57
+
58
+ [tool.ruff]
59
+ line-length = 110
60
+ target-version = "py310"
61
+
62
+ [tool.ruff.lint]
63
+ # Focus on real errors and bugs (pycodestyle errors + pyflakes); skip style modernization.
64
+ select = ["E4", "E7", "E9", "F"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,5 @@
1
+ """PromptSeal β€” regression testing for prompts, agents, and models."""
2
+
3
+ from promptseal._version import __version__
4
+
5
+ __all__ = ["__version__"]
@@ -0,0 +1,3 @@
1
+ from promptseal.cli import app
2
+
3
+ app()
@@ -0,0 +1 @@
1
+ __version__ = "0.2.0"