evalwise 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalwise/__init__.py +23 -0
- evalwise/__main__.py +6 -0
- evalwise/audio/__init__.py +5 -0
- evalwise/audio/assertions.py +335 -0
- evalwise/cli.py +251 -0
- evalwise/core/__init__.py +7 -0
- evalwise/core/context.py +47 -0
- evalwise/core/dataset.py +102 -0
- evalwise/core/result.py +142 -0
- evalwise/core/suite.py +243 -0
- evalwise/image/__init__.py +5 -0
- evalwise/image/assertions.py +382 -0
- evalwise/text/__init__.py +5 -0
- evalwise/text/assertions.py +854 -0
- evalwise-0.1.0.dist-info/METADATA +255 -0
- evalwise-0.1.0.dist-info/RECORD +18 -0
- evalwise-0.1.0.dist-info/WHEEL +4 -0
- evalwise-0.1.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: evalwise
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deterministic-first AI evaluation for text, image, audio, and video
|
|
5
|
+
Project-URL: Homepage, https://github.com/shreyaspj20/evalwise
|
|
6
|
+
Project-URL: Documentation, https://github.com/shreyaspj20/evalwise#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/shreyaspj20/evalwise
|
|
8
|
+
Author-email: Shreyas Reddy <shreyaspj20@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
Keywords: ai,evaluation,llm,machine-learning,ml,testing
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: click>=8.0
|
|
21
|
+
Requires-Dist: httpx>=0.25
|
|
22
|
+
Requires-Dist: jsonschema>=4.0
|
|
23
|
+
Requires-Dist: langdetect>=1.0
|
|
24
|
+
Requires-Dist: pydantic>=2.0
|
|
25
|
+
Requires-Dist: rich>=13.0
|
|
26
|
+
Requires-Dist: textstat>=0.7
|
|
27
|
+
Provides-Extra: all
|
|
28
|
+
Requires-Dist: decord>=0.6; extra == 'all'
|
|
29
|
+
Requires-Dist: librosa>=0.10; extra == 'all'
|
|
30
|
+
Requires-Dist: open-clip-torch>=2.20; extra == 'all'
|
|
31
|
+
Requires-Dist: openai-whisper>=20231117; extra == 'all'
|
|
32
|
+
Requires-Dist: opencv-python>=4.8; extra == 'all'
|
|
33
|
+
Requires-Dist: pillow>=10.0; extra == 'all'
|
|
34
|
+
Requires-Dist: sentence-transformers>=2.0; extra == 'all'
|
|
35
|
+
Requires-Dist: soundfile>=0.12; extra == 'all'
|
|
36
|
+
Requires-Dist: torch>=2.0; extra == 'all'
|
|
37
|
+
Requires-Dist: transformers>=4.35; extra == 'all'
|
|
38
|
+
Requires-Dist: ultralytics>=8.0; extra == 'all'
|
|
39
|
+
Provides-Extra: audio
|
|
40
|
+
Requires-Dist: librosa>=0.10; extra == 'audio'
|
|
41
|
+
Requires-Dist: openai-whisper>=20231117; extra == 'audio'
|
|
42
|
+
Requires-Dist: soundfile>=0.12; extra == 'audio'
|
|
43
|
+
Provides-Extra: dev
|
|
44
|
+
Requires-Dist: mypy>=1.5; extra == 'dev'
|
|
45
|
+
Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
|
|
46
|
+
Requires-Dist: pytest>=7.0; extra == 'dev'
|
|
47
|
+
Requires-Dist: ruff>=0.1; extra == 'dev'
|
|
48
|
+
Provides-Extra: embeddings
|
|
49
|
+
Requires-Dist: sentence-transformers>=2.0; extra == 'embeddings'
|
|
50
|
+
Provides-Extra: image
|
|
51
|
+
Requires-Dist: open-clip-torch>=2.20; extra == 'image'
|
|
52
|
+
Requires-Dist: pillow>=10.0; extra == 'image'
|
|
53
|
+
Requires-Dist: torch>=2.0; extra == 'image'
|
|
54
|
+
Requires-Dist: transformers>=4.35; extra == 'image'
|
|
55
|
+
Requires-Dist: ultralytics>=8.0; extra == 'image'
|
|
56
|
+
Provides-Extra: nli
|
|
57
|
+
Requires-Dist: torch>=2.0; extra == 'nli'
|
|
58
|
+
Requires-Dist: transformers>=4.35; extra == 'nli'
|
|
59
|
+
Provides-Extra: video
|
|
60
|
+
Requires-Dist: decord>=0.6; extra == 'video'
|
|
61
|
+
Requires-Dist: opencv-python>=4.8; extra == 'video'
|
|
62
|
+
Description-Content-Type: text/markdown
|
|
63
|
+
|
|
64
|
+
# EvalWise
|
|
65
|
+
|
|
66
|
+
**Deterministic-first AI evaluation for text, image, audio, and video.**
|
|
67
|
+
|
|
68
|
+
The eval SDK that doesn't default to "ask another AI if this is good."
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
pip install evalwise
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Why EvalWise?
|
|
75
|
+
|
|
76
|
+
Most eval tools jump straight to LLM-as-judge. That's:
|
|
77
|
+
- **Expensive** — every eval is another API call
|
|
78
|
+
- **Non-deterministic** — same input, different scores
|
|
79
|
+
- **Ungrounded** — "Score: 4/5" tells you nothing
|
|
80
|
+
|
|
81
|
+
EvalWise flips the default: **deterministic checks first, LLM-judge only when you have to.**
|
|
82
|
+
|
|
83
|
+
## Quick Start
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
from evalwise import Suite, Assert
|
|
87
|
+
|
|
88
|
+
suite = Suite("summarization")
|
|
89
|
+
|
|
90
|
+
@suite.test
|
|
91
|
+
def test_format(response: str):
|
|
92
|
+
Assert.bullet_count(response, exactly=3)
|
|
93
|
+
Assert.word_count(response, max=200)
|
|
94
|
+
Assert.json_valid(response)
|
|
95
|
+
|
|
96
|
+
@suite.test
|
|
97
|
+
def test_factuality(response: str, source: str):
|
|
98
|
+
Assert.entails(response, source=source)
|
|
99
|
+
Assert.no_contradiction(response, source=source)
|
|
100
|
+
|
|
101
|
+
# Run against a dataset
|
|
102
|
+
results = suite.run(dataset="./golden_set.json")
|
|
103
|
+
results.assert_pass_rate(threshold=0.95)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## CLI
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
# Run eval suite
|
|
110
|
+
evalwise run tests/test_summary.py --dataset golden.json
|
|
111
|
+
|
|
112
|
+
# Run in CI (exit code 1 on failure)
|
|
113
|
+
evalwise run tests/ --ci --threshold 0.95
|
|
114
|
+
|
|
115
|
+
# Create sample eval file
|
|
116
|
+
evalwise init
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
## Text Assertions
|
|
120
|
+
|
|
121
|
+
| Assertion | What it checks |
|
|
122
|
+
|-----------|----------------|
|
|
123
|
+
| `contains(text, substring)` | Substring present |
|
|
124
|
+
| `not_contains(text, substring)` | Substring absent |
|
|
125
|
+
| `regex(text, pattern)` | Pattern match |
|
|
126
|
+
| `json_valid(text)` | Parseable JSON |
|
|
127
|
+
| `json_schema(text, schema)` | Matches JSON schema |
|
|
128
|
+
| `word_count(text, min, max)` | Word count in range |
|
|
129
|
+
| `bullet_count(text, exactly)` | Bullet point count |
|
|
130
|
+
| `readability(text, min_score)` | Flesch Reading Ease |
|
|
131
|
+
| `language_is(text, "en")` | Correct language |
|
|
132
|
+
| `code_parses(text, "python")` | Valid syntax |
|
|
133
|
+
| `code_runs(text)` | Executes without error |
|
|
134
|
+
| `entails(text, source)` | Follows from source (NLI) |
|
|
135
|
+
| `no_contradiction(text, source)` | No contradictions |
|
|
136
|
+
| `embedding_similarity(text, ref)` | Semantic similarity |
|
|
137
|
+
| `urls_valid(text)` | All URLs return 2xx |
|
|
138
|
+
|
|
139
|
+
## Image Assertions
|
|
140
|
+
|
|
141
|
+
Image evals cover prompt alignment, object detection, resolution/aspect ratio, NSFW safety, and similarity. Object detection uses **YOLO** via `ultralytics`.
|
|
142
|
+
|
|
143
|
+
```python
|
|
144
|
+
from evalwise.image import ImageAssert
|
|
145
|
+
|
|
146
|
+
# Prompt alignment with CLIP
|
|
147
|
+
ImageAssert.clip_score(image, "a cat on a couch", threshold=0.25)
|
|
148
|
+
|
|
149
|
+
# Object detection
|
|
150
|
+
ImageAssert.contains_object(image, "cat", confidence=0.5)
|
|
151
|
+
ImageAssert.object_count(image, "person", exactly=2, confidence=0.5)
|
|
152
|
+
|
|
153
|
+
# Metadata
|
|
154
|
+
ImageAssert.resolution_is(image, width=1024, height=1024)
|
|
155
|
+
ImageAssert.resolution_min(image, width=512, height=512)
|
|
156
|
+
ImageAssert.aspect_ratio(image, ratio=1.0, tolerance=0.1)
|
|
157
|
+
ImageAssert.format_is(image, "PNG")
|
|
158
|
+
|
|
159
|
+
# Safety
|
|
160
|
+
ImageAssert.nsfw_below(image, threshold=0.1)
|
|
161
|
+
|
|
162
|
+
# Similarity to a reference image
|
|
163
|
+
ImageAssert.image_similarity(image, reference, threshold=0.8)
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
### Image generation example
|
|
167
|
+
|
|
168
|
+
`examples/test_image_generation.py` shows a complete eval suite. The dataset can include per-image thresholds and object lists:
|
|
169
|
+
|
|
170
|
+
```json
|
|
171
|
+
{
|
|
172
|
+
"image_path": "./examples/sample_cat.jpeg",
|
|
173
|
+
"prompt": "a person sitting on a couch with a dog and a cat",
|
|
174
|
+
"min_width": 200,
|
|
175
|
+
"min_height": 100,
|
|
176
|
+
"required_objects": ["dog"],
|
|
177
|
+
"confidence": 0.25
|
|
178
|
+
}
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
## Audio Assertions
|
|
182
|
+
|
|
183
|
+
Audio assertions that use Whisper (`transcription_contains`, `transcription_equals`, `language_is`) require **ffmpeg** to be installed on your system in addition to the `evalwise[audio]` Python dependencies.
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from evalwise.audio import AudioAssert
|
|
187
|
+
|
|
188
|
+
AudioAssert.transcription_contains(audio, "hello world")
|
|
189
|
+
AudioAssert.transcription_equals(audio, expected_text)
|
|
190
|
+
AudioAssert.language_is(audio, "en")
|
|
191
|
+
AudioAssert.duration_between(audio, min_sec=5, max_sec=30)
|
|
192
|
+
AudioAssert.sample_rate_is(audio, hz=44100)
|
|
193
|
+
AudioAssert.no_silence(audio, max_silence_sec=1.0)
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
## Installation
|
|
197
|
+
|
|
198
|
+
```bash
|
|
199
|
+
# Core (text assertions)
|
|
200
|
+
pip install evalwise
|
|
201
|
+
|
|
202
|
+
# With image support
|
|
203
|
+
pip install evalwise[image]
|
|
204
|
+
|
|
205
|
+
# With audio support (also requires ffmpeg system binary)
|
|
206
|
+
pip install evalwise[audio]
|
|
207
|
+
|
|
208
|
+
# On macOS: brew install ffmpeg
|
|
209
|
+
# On Ubuntu: sudo apt install ffmpeg
|
|
210
|
+
# On Windows: winget install Gyan.FFmpeg
|
|
211
|
+
|
|
212
|
+
# Everything
|
|
213
|
+
pip install evalwise[all]
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
### What each extra gives you
|
|
217
|
+
|
|
218
|
+
| Extra | Capabilities enabled | Example tests |
|
|
219
|
+
|---|---|---|
|
|
220
|
+
| *(none)* | 22 of 25 text assertions | `test_basic.py` |
|
|
221
|
+
| `[nli]` | `Assert.entails`, `Assert.no_contradiction` | `test_summarization.py` |
|
|
222
|
+
| `[embeddings]` | `Assert.embedding_similarity` | — |
|
|
223
|
+
| `[image]` | CLIP, YOLO object detection, NSFW, resolution, similarity | `test_image_generation.py` |
|
|
224
|
+
| `[audio]` | Whisper transcription, language, duration, sample rate | `test_audio.py` |
|
|
225
|
+
| `[video]` | Video assertions (extra only, no example yet) | — |
|
|
226
|
+
| `[all]` | All of the above | — |
|
|
227
|
+
|
|
228
|
+
### System dependencies
|
|
229
|
+
|
|
230
|
+
Some examples also require system binaries:
|
|
231
|
+
|
|
232
|
+
| Capability | System binary | Install |
|
|
233
|
+
|---|---|---|
|
|
234
|
+
| Audio transcription | `ffmpeg` | `brew install ffmpeg` / `apt install ffmpeg` |
|
|
235
|
+
|
|
236
|
+
## Philosophy
|
|
237
|
+
|
|
238
|
+
1. **Deterministic by default** — Same input, same result. Always.
|
|
239
|
+
2. **Cheap first** — Check format, length, syntax before calling models.
|
|
240
|
+
3. **Grounded scores** — Know exactly why something failed.
|
|
241
|
+
4. **LLM-judge as last resort** — Only for truly subjective criteria.
|
|
242
|
+
|
|
243
|
+
## Comparison
|
|
244
|
+
|
|
245
|
+
| | EvalWise | Promptfoo | Braintrust | LangSmith |
|
|
246
|
+
|---|--------|-----------|------------|-----------|
|
|
247
|
+
| Deterministic-first | ✓ | Partial | ✗ | ✗ |
|
|
248
|
+
| Image/Audio evals | ✓ | ✗ | ✗ | ✗ |
|
|
249
|
+
| Python-native | ✓ | YAML | SDK | SDK |
|
|
250
|
+
| No account required | ✓ | ✓ | ✗ | ✗ |
|
|
251
|
+
| OSS | ✓ | ✓ | Partial | ✗ |
|
|
252
|
+
|
|
253
|
+
## License
|
|
254
|
+
|
|
255
|
+
MIT
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
evalwise/__init__.py,sha256=se25ej6Q09D6IHa6H3oFgcFa2CLPj8fW2_wFALt7_JE,521
|
|
2
|
+
evalwise/__main__.py,sha256=O-ufHkaXo66k4JJuvtq-LLohfHTegw97iRZp5WIMTII,115
|
|
3
|
+
evalwise/cli.py,sha256=maRRYw68N0r1jaF45bvbX9hioSPHsQUekyxQrpGPGxA,7663
|
|
4
|
+
evalwise/audio/__init__.py,sha256=C1i2dKZKbv6isTDlE9n7dgBnSidEcTSsXVsvyuBQ54o,151
|
|
5
|
+
evalwise/audio/assertions.py,sha256=Be0pgt7TIZsnRWOBdvfBwGOXDd_qGqqgcKo28iAUCec,11254
|
|
6
|
+
evalwise/core/__init__.py,sha256=vREc5gtO5fzZ7hZN77GPE8EXmauteBoW5A6AzGzJiF4,268
|
|
7
|
+
evalwise/core/context.py,sha256=x_rGN3Mq0gMMUmY5p0TppqOUaPonQ8wfSxFlhUDy4Bs,1285
|
|
8
|
+
evalwise/core/dataset.py,sha256=O8G7xq7rxFYFFF8rUZcjIiX5TxJzdT1xiAFYph9oBsQ,3066
|
|
9
|
+
evalwise/core/result.py,sha256=9Zap5uwgSx4ZEpo7JWIlmMisECfJz_RnBiFmXWPBki4,4142
|
|
10
|
+
evalwise/core/suite.py,sha256=ZSch2WvjnsUI_Pkp_G6Bs94lVolIeaiHr8IJxwBrllc,7407
|
|
11
|
+
evalwise/image/__init__.py,sha256=4tyIocTBpdTAzzowzAY85rSVymCKHwtLRXVjY0chm_k,152
|
|
12
|
+
evalwise/image/assertions.py,sha256=M36067ue6qWpVrCh6XXBmpSsMN5va2wQgM3PaV5pBSE,13213
|
|
13
|
+
evalwise/text/__init__.py,sha256=uG65z9-FE7CPaSBWaXT1Jgf8hRApS4Fp3ESa5N283kI,140
|
|
14
|
+
evalwise/text/assertions.py,sha256=8NrF0PHLZR7LDxPheBBUlF24Wr89HebR_4NCdkFK7vo,29208
|
|
15
|
+
evalwise-0.1.0.dist-info/METADATA,sha256=_xmWJqRJOqxSCkSyCiwyGFVn6aliSg9et1xvzW9oasQ,8423
|
|
16
|
+
evalwise-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
17
|
+
evalwise-0.1.0.dist-info/entry_points.txt,sha256=h4a9iwz0QrAkN0jCjU4LW84KtesnIGwe0TgT3ETNGWU,47
|
|
18
|
+
evalwise-0.1.0.dist-info/RECORD,,
|