mic-evals 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. mic_evals-0.1.0/.gitignore +12 -0
  2. mic_evals-0.1.0/.python-version +1 -0
  3. mic_evals-0.1.0/LICENSE +21 -0
  4. mic_evals-0.1.0/PKG-INFO +268 -0
  5. mic_evals-0.1.0/README.md +243 -0
  6. mic_evals-0.1.0/docs/api.md +125 -0
  7. mic_evals-0.1.0/docs/artifact-schemas/case-record-v2.schema.json +87 -0
  8. mic_evals-0.1.0/docs/artifact-schemas/common-v2.schema.json +418 -0
  9. mic_evals-0.1.0/docs/artifact-schemas/dataset-row-v2.schema.json +22 -0
  10. mic_evals-0.1.0/docs/artifact-schemas/run-v2.schema.json +332 -0
  11. mic_evals-0.1.0/docs/artifacts.md +199 -0
  12. mic_evals-0.1.0/docs/providers.md +262 -0
  13. mic_evals-0.1.0/docs/releasing.md +67 -0
  14. mic_evals-0.1.0/docs/reporting.md +187 -0
  15. mic_evals-0.1.0/docs/schemas.md +255 -0
  16. mic_evals-0.1.0/docs/verification.md +128 -0
  17. mic_evals-0.1.0/examples/__init__.py +1 -0
  18. mic_evals-0.1.0/examples/async_eval.py +34 -0
  19. mic_evals-0.1.0/examples/failures.py +54 -0
  20. mic_evals-0.1.0/examples/fixtures/structured.jsonl +2 -0
  21. mic_evals-0.1.0/examples/fixtures/tickets.jsonl +3 -0
  22. mic_evals-0.1.0/examples/fixtures/triage.jsonl +3 -0
  23. mic_evals-0.1.0/examples/provider_equivalence.py +69 -0
  24. mic_evals-0.1.0/examples/pydantic_models.py +33 -0
  25. mic_evals-0.1.0/examples/structured.py +116 -0
  26. mic_evals-0.1.0/examples/ticket_eval.py +83 -0
  27. mic_evals-0.1.0/examples/triage.py +53 -0
  28. mic_evals-0.1.0/pyproject.toml +80 -0
  29. mic_evals-0.1.0/scripts/check_coverage.py +48 -0
  30. mic_evals-0.1.0/scripts/verify.py +148 -0
  31. mic_evals-0.1.0/scripts/verify_packaging.py +348 -0
  32. mic_evals-0.1.0/scripts/verify_release.py +120 -0
  33. mic_evals-0.1.0/scripts/verify_typing.py +54 -0
  34. mic_evals-0.1.0/src/mic/__init__.py +59 -0
  35. mic_evals-0.1.0/src/mic/__main__.py +3 -0
  36. mic_evals-0.1.0/src/mic/_async.py +36 -0
  37. mic_evals-0.1.0/src/mic/_runtime/__init__.py +1 -0
  38. mic_evals-0.1.0/src/mic/_runtime/artifacts.py +177 -0
  39. mic_evals-0.1.0/src/mic/_runtime/batch.py +78 -0
  40. mic_evals-0.1.0/src/mic/_runtime/callbacks.py +51 -0
  41. mic_evals-0.1.0/src/mic/_runtime/case.py +111 -0
  42. mic_evals-0.1.0/src/mic/_runtime/contracts.py +14 -0
  43. mic_evals-0.1.0/src/mic/_runtime/discovery.py +57 -0
  44. mic_evals-0.1.0/src/mic/_runtime/engine.py +279 -0
  45. mic_evals-0.1.0/src/mic/_runtime/files.py +27 -0
  46. mic_evals-0.1.0/src/mic/_runtime/materialization.py +222 -0
  47. mic_evals-0.1.0/src/mic/_runtime/options.py +76 -0
  48. mic_evals-0.1.0/src/mic/_runtime/reporting.py +68 -0
  49. mic_evals-0.1.0/src/mic/_runtime/summary.py +107 -0
  50. mic_evals-0.1.0/src/mic/_runtime/validation.py +140 -0
  51. mic_evals-0.1.0/src/mic/cli.py +216 -0
  52. mic_evals-0.1.0/src/mic/decorators.py +130 -0
  53. mic_evals-0.1.0/src/mic/errors.py +17 -0
  54. mic_evals-0.1.0/src/mic/integrations/__init__.py +3 -0
  55. mic_evals-0.1.0/src/mic/integrations/pydantic.py +120 -0
  56. mic_evals-0.1.0/src/mic/models.py +152 -0
  57. mic_evals-0.1.0/src/mic/providers/__init__.py +21 -0
  58. mic_evals-0.1.0/src/mic/providers/_braintrust/__init__.py +1 -0
  59. mic_evals-0.1.0/src/mic/providers/_braintrust/transport.py +184 -0
  60. mic_evals-0.1.0/src/mic/providers/_io.py +98 -0
  61. mic_evals-0.1.0/src/mic/providers/base.py +102 -0
  62. mic_evals-0.1.0/src/mic/providers/bigquery.py +328 -0
  63. mic_evals-0.1.0/src/mic/providers/braintrust.py +248 -0
  64. mic_evals-0.1.0/src/mic/providers/files.py +114 -0
  65. mic_evals-0.1.0/src/mic/providers/memory.py +78 -0
  66. mic_evals-0.1.0/src/mic/py.typed +0 -0
  67. mic_evals-0.1.0/src/mic/reporters/__init__.py +7 -0
  68. mic_evals-0.1.0/src/mic/reporters/base.py +15 -0
  69. mic_evals-0.1.0/src/mic/reporters/braintrust.py +229 -0
  70. mic_evals-0.1.0/src/mic/reporters/console.py +47 -0
  71. mic_evals-0.1.0/src/mic/reporters/html.py +82 -0
  72. mic_evals-0.1.0/src/mic/reporters/templates/report.html +21 -0
  73. mic_evals-0.1.0/src/mic/reporters/templates/report.js +219 -0
  74. mic_evals-0.1.0/src/mic/runner.py +5 -0
  75. mic_evals-0.1.0/src/mic/schema/__init__.py +64 -0
  76. mic_evals-0.1.0/src/mic/schema/_compiler.py +176 -0
  77. mic_evals-0.1.0/src/mic/schema/_contracts.py +76 -0
  78. mic_evals-0.1.0/src/mic/schema/_dataclasses.py +148 -0
  79. mic_evals-0.1.0/src/mic/schema/_values.py +236 -0
  80. mic_evals-0.1.0/tests/__init__.py +1 -0
  81. mic_evals-0.1.0/tests/acceptance/test_cli.py +189 -0
  82. mic_evals-0.1.0/tests/acceptance/test_cli_errors.py +85 -0
  83. mic_evals-0.1.0/tests/acceptance/test_readme.py +166 -0
  84. mic_evals-0.1.0/tests/datasets/__init__.py +1 -0
  85. mic_evals-0.1.0/tests/datasets/test_identity_and_cleanup.py +115 -0
  86. mic_evals-0.1.0/tests/datasets/test_materialization.py +310 -0
  87. mic_evals-0.1.0/tests/integration/test_live_providers.py +71 -0
  88. mic_evals-0.1.0/tests/packaging/__init__.py +1 -0
  89. mic_evals-0.1.0/tests/packaging/test_core_imports.py +48 -0
  90. mic_evals-0.1.0/tests/packaging/test_dependencies.py +80 -0
  91. mic_evals-0.1.0/tests/packaging/test_public_api.py +63 -0
  92. mic_evals-0.1.0/tests/packaging/test_release.py +172 -0
  93. mic_evals-0.1.0/tests/providers/__init__.py +1 -0
  94. mic_evals-0.1.0/tests/providers/test_bigquery_provider.py +222 -0
  95. mic_evals-0.1.0/tests/providers/test_bigquery_sdk_transport.py +206 -0
  96. mic_evals-0.1.0/tests/providers/test_braintrust_provider.py +244 -0
  97. mic_evals-0.1.0/tests/providers/test_braintrust_sdk_transport.py +311 -0
  98. mic_evals-0.1.0/tests/providers/test_file_memory_providers.py +166 -0
  99. mic_evals-0.1.0/tests/providers/test_provider_equivalence.py +219 -0
  100. mic_evals-0.1.0/tests/reporting/__init__.py +1 -0
  101. mic_evals-0.1.0/tests/reporting/_fixtures.py +45 -0
  102. mic_evals-0.1.0/tests/reporting/js/package-lock.json +546 -0
  103. mic_evals-0.1.0/tests/reporting/js/package.json +8 -0
  104. mic_evals-0.1.0/tests/reporting/js/report.test.mjs +230 -0
  105. mic_evals-0.1.0/tests/reporting/test_braintrust.py +352 -0
  106. mic_evals-0.1.0/tests/reporting/test_html.py +149 -0
  107. mic_evals-0.1.0/tests/runtime/__init__.py +1 -0
  108. mic_evals-0.1.0/tests/runtime/artifact_contract.py +54 -0
  109. mic_evals-0.1.0/tests/runtime/helpers.py +33 -0
  110. mic_evals-0.1.0/tests/runtime/test_artifact_contract.py +264 -0
  111. mic_evals-0.1.0/tests/runtime/test_artifacts.py +41 -0
  112. mic_evals-0.1.0/tests/runtime/test_cancellation.py +252 -0
  113. mic_evals-0.1.0/tests/runtime/test_cases.py +202 -0
  114. mic_evals-0.1.0/tests/runtime/test_finalization.py +341 -0
  115. mic_evals-0.1.0/tests/runtime/test_options.py +148 -0
  116. mic_evals-0.1.0/tests/runtime/test_scheduling.py +125 -0
  117. mic_evals-0.1.0/tests/runtime/test_setup_failures.py +109 -0
  118. mic_evals-0.1.0/tests/runtime/test_statistics.py +69 -0
  119. mic_evals-0.1.0/tests/runtime/test_structured.py +143 -0
  120. mic_evals-0.1.0/tests/schema/__init__.py +1 -0
  121. mic_evals-0.1.0/tests/schema/_fixtures.py +40 -0
  122. mic_evals-0.1.0/tests/schema/test_dataclasses.py +244 -0
  123. mic_evals-0.1.0/tests/schema/test_pydantic.py +206 -0
  124. mic_evals-0.1.0/tests/schema/test_values.py +195 -0
  125. mic_evals-0.1.0/tests/typing/negative.py +36 -0
  126. mic_evals-0.1.0/tests/typing/positive.py +35 -0
  127. mic_evals-0.1.0/tests/typing/pyrightconfig.json +17 -0
  128. mic_evals-0.1.0/tests/typing/structured.py +27 -0
  129. mic_evals-0.1.0/tests/verification/test_coverage.py +87 -0
  130. mic_evals-0.1.0/uv.lock +1575 -0
@@ -0,0 +1,12 @@
1
+ .venv/
2
+ .cache/
3
+ .artifacts/
4
+ .pytest_cache/
5
+ .ruff_cache/
6
+ __pycache__/
7
+ *.pyc
8
+ .mic/
9
+ dist/
10
+ *.egg-info/
11
+ node_modules/
12
+ .coverage
@@ -0,0 +1 @@
1
+ 3.12
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ryan Eiger
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,268 @@
1
+ Metadata-Version: 2.5
2
+ Name: mic-evals
3
+ Version: 0.1.0
4
+ Summary: Typed, provider-independent micro-evaluations for Python
5
+ Project-URL: Homepage, https://github.com/rybosome/mic
6
+ Project-URL: Repository, https://github.com/rybosome/mic
7
+ Project-URL: Issues, https://github.com/rybosome/mic/issues
8
+ Author: Ryan Eiger
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: evals,evaluation,llm,testing
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Typing :: Typed
17
+ Requires-Python: >=3.12
18
+ Provides-Extra: bigquery
19
+ Requires-Dist: google-cloud-bigquery<4,>=3.30; extra == 'bigquery'
20
+ Provides-Extra: braintrust
21
+ Requires-Dist: braintrust==0.39.0; extra == 'braintrust'
22
+ Provides-Extra: pydantic
23
+ Requires-Dist: pydantic<3,>=2.11; extra == 'pydantic'
24
+ Description-Content-Type: text/markdown
25
+
26
+ # mic
27
+
28
+ [![CI](https://github.com/rybosome/mic/actions/workflows/ci.yml/badge.svg)](https://github.com/rybosome/mic/actions/workflows/ci.yml)
29
+
30
+ **Did that prompt, model, or application change actually improve the answers?**
31
+
32
+ Mic is a small Python evaluation harness for systems whose answers can vary.
33
+ Give it examples, the code you want to evaluate, and a way to score the results.
34
+ Run the evaluation repeatedly, then inspect what happened, case by case.
35
+
36
+ Use it to test an LLM classifier, score an agent's work, or evaluate another ML
37
+ application. You keep your application code and choose your own models and scoring
38
+ logic; Mic handles dataset validation, bounded concurrent execution, repeated
39
+ trials, and local evidence. No hosted evaluation platform is required.
40
+
41
+ ## Evaluate a support-ticket classifier
42
+
43
+ Suppose you're using an LLM to sort support messages into bugs, feature requests,
44
+ and questions. Before changing its prompt or model, give yourself a repeatable check.
45
+
46
+ With Python 3.12+, install Mic and the SDK used by this example:
47
+
48
+ ```console
49
+ python -m pip install "mic-evals[pydantic]" openai
50
+ ```
51
+
52
+ Mic's core has no third-party runtime dependencies. This example opts into Pydantic
53
+ and the OpenAI SDK to share a typed output contract between Mic and the model call.
54
+ It uses [structured outputs](https://developers.openai.com/api/docs/guides/structured-outputs)
55
+ with [GPT-4.1 mini](https://developers.openai.com/api/docs/models/gpt-4.1-mini);
56
+ replace the classifier function with your own model or application call.
57
+ Set `OPENAI_API_KEY` in your environment using your usual secret-management method.
58
+ Running it sends the example messages to OpenAI and incurs normal API charges.
59
+
60
+ Save this complete example as `ticket_eval.py`:
61
+
62
+ ```python
63
+ from typing import Literal
64
+
65
+ from pydantic import BaseModel
66
+
67
+ import mic
68
+
69
+ ##
70
+ ## Define a dataset
71
+ ##
72
+
73
+
74
+ class Ticket(BaseModel):
75
+ subject: str
76
+ body: str
77
+
78
+
79
+ class Classification(BaseModel):
80
+ label: Literal["bug", "feature", "question"]
81
+
82
+
83
+ @mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
84
+ def tickets() -> list[mic.RawCase]:
85
+ return [
86
+ mic.RawCase(
87
+ id="upload",
88
+ input=Ticket(subject="PDF upload", body="The app closes whenever I upload a PDF."),
89
+ expected=Classification(label="bug"),
90
+ ),
91
+ mic.RawCase(
92
+ id="export",
93
+ input=Ticket(
94
+ subject="Invoice export", body="Can you add an option to export invoices?"
95
+ ),
96
+ expected=Classification(label="feature"),
97
+ ),
98
+ mic.RawCase(
99
+ id="invoice",
100
+ input=Ticket(subject="Past invoice", body="Where can I download last month's invoice?"),
101
+ expected=Classification(label="question"),
102
+ ),
103
+ ]
104
+
105
+
106
+ ##
107
+ ## Define scoring
108
+ ##
109
+
110
+
111
+ @mic.scorer(name="accuracy", requires_expected=True)
112
+ def accuracy(
113
+ ctx: mic.ScoreContext[Ticket, Classification, Classification, mic.JsonObject],
114
+ ) -> float:
115
+ return float(ctx.output.label == ctx.require_expected().label)
116
+
117
+
118
+ ##
119
+ ## Define the task
120
+ ##
121
+
122
+
123
+ @mic.eval(name="classify", dataset=tickets, output=Classification, scorers=[accuracy])
124
+ def classify(
125
+ ctx: mic.TaskContext[Classification, mic.JsonObject], ticket: Ticket
126
+ ) -> Classification:
127
+ from openai import OpenAI
128
+
129
+ # Create the client only when the task runs, and close it after the call.
130
+ with OpenAI(timeout=30, max_retries=0) as client:
131
+ response = client.responses.parse(
132
+ model="gpt-4.1-mini",
133
+ instructions=(
134
+ "Classify the support ticket as "
135
+ "bug (broken behavior), feature (new capability), or question (how-to)."
136
+ ),
137
+ input=ticket.model_dump_json(),
138
+ text_format=Classification,
139
+ store=False,
140
+ )
141
+ if response.output_parsed is None:
142
+ raise ValueError("The model did not return a classification.")
143
+ return response.output_parsed
144
+ ```
145
+
146
+ `Ticket` gives the task typed inputs; `Classification` defines both the expected
147
+ answer and the model's structured output. The SDK derives its output schema from
148
+ that class and parses the response into it. Mic validates dataset values and task
149
+ outputs against the same types, so the scorer works with objects, not JSON parsing
150
+ or string cleanup.
151
+
152
+ A valid but wrong label scores `0`; the right label scores `1`. Invalid output or
153
+ no parsed classification (for example, a refusal) is an execution failure, not a
154
+ wrong answer. These three cases are illustrative; a useful evaluation needs a
155
+ larger, representative set of labeled tickets.
156
+
157
+ Run it from the directory containing `ticket_eval.py`, then open the report:
158
+
159
+ ```console
160
+ mic run ticket_eval:classify --output .mic/tickets-first
161
+ mic report .mic/tickets-first --open
162
+ ```
163
+
164
+ That's three classifier calls. The report shows each message, its expected label,
165
+ the actual response, and its score, alongside aggregate accuracy and execution
166
+ failures. Review low-scoring cases for wrong answers or ambiguous expected labels,
167
+ and execution failures for calls that did not produce a valid classification.
168
+
169
+ Each run saves its dataset snapshot, per-trial results, and summary as JSON/JSONL,
170
+ plus a self-contained HTML report you can open without a server or network.
171
+ Explicit output directories must be empty; omit `--output` to get a unique directory
172
+ automatically. A successfully executed run can still have poor scores—execution
173
+ success and answer quality are separate.
174
+
175
+ The same code lives in [examples/ticket_eval.py](examples/ticket_eval.py). To explore
176
+ the report without credentials or API charges, the repository also includes an
177
+ [offline classification demo](examples/triage.py) with deliberately imperfect rules;
178
+ see the [walkthrough](docs/verification.md).
179
+
180
+ ## Repeat, check, improve
181
+
182
+ ### Look for variation
183
+
184
+ Run each message five times, with at most two tasks executing concurrently:
185
+
186
+ ```console
187
+ mic run ticket_eval:classify --trials 5 --concurrency 2 --output .mic/tickets-repeat
188
+ ```
189
+
190
+ This makes 15 classifier calls. Inspect individual trials as well as the mean:
191
+ one message that fails intermittently deserves attention even if the average looks
192
+ good. Repetition gives you more observations, not proof of statistical significance.
193
+
194
+ Change the prompt or model in `classify`, run again into a fresh directory,
195
+ and review both reports against the same labeled cases. Keep the evaluation set
196
+ representative rather than tuning only to these three examples.
197
+
198
+ ### Make quality a check
199
+
200
+ Add a score requirement when you're ready to use the evaluation locally or in CI:
201
+
202
+ ```console
203
+ mic run ticket_eval:classify --trials 5 --require 'accuracy>=0.9'
204
+ ```
205
+
206
+ The command exits nonzero if execution fails or mean accuracy falls below `0.9`.
207
+ That threshold is illustrative; choose one appropriate to your dataset and the cost
208
+ of a wrong answer. A failed quality gate still leaves local results to investigate.
209
+
210
+ ### Grow the dataset without changing the task
211
+
212
+ Save the same cases as `tickets.jsonl` beside `ticket_eval.py`:
213
+
214
+ ```jsonl
215
+ {"id":"upload","input":{"subject":"PDF upload","body":"The app closes whenever I upload a PDF."},"expected":{"label":"bug"}}
216
+ {"id":"export","input":{"subject":"Invoice export","body":"Can you add an option to export invoices?"},"expected":{"label":"feature"}}
217
+ {"id":"invoice","input":{"subject":"Past invoice","body":"Where can I download last month's invoice?"},"expected":{"label":"question"}}
218
+ ```
219
+
220
+ Replace the `tickets` definition with this, adding the two imports:
221
+
222
+ ```python
223
+ from pathlib import Path
224
+
225
+ from mic.providers.files import FileHandle
226
+
227
+
228
+ @mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
229
+ def tickets() -> FileHandle:
230
+ return FileHandle(Path(__file__).with_name("tickets.jsonl"))
231
+ ```
232
+
233
+ Mic hydrates the JSON objects into `Ticket` and `Classification` instances and
234
+ rejects invalid cases before calling the model. The task, scorer, and run command
235
+ stay the same. Add cases from real failures as
236
+ you encounter them—for example, a message that sounds like a feature request but
237
+ describes an existing feature that stopped working.
238
+
239
+ ## Bring your own application
240
+
241
+ The same pattern applies beyond classification: score extracted fields, check an
242
+ agent's result against a rubric, or compute a metric for an ML prediction. Tasks
243
+ and scorers are Python functions, so they can call your existing code. Sync and
244
+ async functions are supported; scripts can use `mic.run()` and notebooks can use
245
+ `await mic.arun()` instead of the CLI.
246
+
247
+ - **Structured data:** use ordinary dataclasses for inputs, outputs, and expected
248
+ values; see the [structured example](examples/structured.py) and [schema guide](docs/schemas.md).
249
+ - **Other dataset sources:** load local files, BigQuery queries, or versioned
250
+ Braintrust datasets with optional integrations; see [providers](docs/providers.md).
251
+ - **Remote reporting:** optionally export results to a Braintrust experiment after
252
+ saving local evidence. Dataset storage and reporting are independent; see [reporting](docs/reporting.md).
253
+
254
+ ## Before you use it
255
+
256
+ Mic is an early release, and its public API may change before a stable release.
257
+ It runs finite, bounded datasets locally—not distributed jobs or an application
258
+ hosting service. Synchronous callbacks must finish cooperatively: a timeout cannot
259
+ forcibly stop a running Python thread.
260
+
261
+ Reports and artifacts contain your actual evaluation data, including inputs and
262
+ outputs. They are **not automatically redacted**. Treat them as sensitive, keep
263
+ credentials out of your cases, and review evidence before sharing or committing it.
264
+ Your model calls and optional cloud integrations have their own costs and data-handling
265
+ policies. See [artifact handling](docs/artifacts.md) for details.
266
+
267
+ [API reference](docs/api.md) · [Verification](docs/verification.md) ·
268
+ [Contributing](CONTRIBUTING.md) · [Releasing](docs/releasing.md) · [MIT license](LICENSE)
@@ -0,0 +1,243 @@
1
+ # mic
2
+
3
+ [![CI](https://github.com/rybosome/mic/actions/workflows/ci.yml/badge.svg)](https://github.com/rybosome/mic/actions/workflows/ci.yml)
4
+
5
+ **Did that prompt, model, or application change actually improve the answers?**
6
+
7
+ Mic is a small Python evaluation harness for systems whose answers can vary.
8
+ Give it examples, the code you want to evaluate, and a way to score the results.
9
+ Run the evaluation repeatedly, then inspect what happened, case by case.
10
+
11
+ Use it to test an LLM classifier, score an agent's work, or evaluate another ML
12
+ application. You keep your application code and choose your own models and scoring
13
+ logic; Mic handles dataset validation, bounded concurrent execution, repeated
14
+ trials, and local evidence. No hosted evaluation platform is required.
15
+
16
+ ## Evaluate a support-ticket classifier
17
+
18
+ Suppose you're using an LLM to sort support messages into bugs, feature requests,
19
+ and questions. Before changing its prompt or model, give yourself a repeatable check.
20
+
21
+ With Python 3.12+, install Mic and the SDK used by this example:
22
+
23
+ ```console
24
+ python -m pip install "mic-evals[pydantic]" openai
25
+ ```
26
+
27
+ Mic's core has no third-party runtime dependencies. This example opts into Pydantic
28
+ and the OpenAI SDK to share a typed output contract between Mic and the model call.
29
+ It uses [structured outputs](https://developers.openai.com/api/docs/guides/structured-outputs)
30
+ with [GPT-4.1 mini](https://developers.openai.com/api/docs/models/gpt-4.1-mini);
31
+ replace the classifier function with your own model or application call.
32
+ Set `OPENAI_API_KEY` in your environment using your usual secret-management method.
33
+ Running it sends the example messages to OpenAI and incurs normal API charges.
34
+
35
+ Save this complete example as `ticket_eval.py`:
36
+
37
+ ```python
38
+ from typing import Literal
39
+
40
+ from pydantic import BaseModel
41
+
42
+ import mic
43
+
44
+ ##
45
+ ## Define a dataset
46
+ ##
47
+
48
+
49
+ class Ticket(BaseModel):
50
+ subject: str
51
+ body: str
52
+
53
+
54
+ class Classification(BaseModel):
55
+ label: Literal["bug", "feature", "question"]
56
+
57
+
58
+ @mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
59
+ def tickets() -> list[mic.RawCase]:
60
+ return [
61
+ mic.RawCase(
62
+ id="upload",
63
+ input=Ticket(subject="PDF upload", body="The app closes whenever I upload a PDF."),
64
+ expected=Classification(label="bug"),
65
+ ),
66
+ mic.RawCase(
67
+ id="export",
68
+ input=Ticket(
69
+ subject="Invoice export", body="Can you add an option to export invoices?"
70
+ ),
71
+ expected=Classification(label="feature"),
72
+ ),
73
+ mic.RawCase(
74
+ id="invoice",
75
+ input=Ticket(subject="Past invoice", body="Where can I download last month's invoice?"),
76
+ expected=Classification(label="question"),
77
+ ),
78
+ ]
79
+
80
+
81
+ ##
82
+ ## Define scoring
83
+ ##
84
+
85
+
86
+ @mic.scorer(name="accuracy", requires_expected=True)
87
+ def accuracy(
88
+ ctx: mic.ScoreContext[Ticket, Classification, Classification, mic.JsonObject],
89
+ ) -> float:
90
+ return float(ctx.output.label == ctx.require_expected().label)
91
+
92
+
93
+ ##
94
+ ## Define the task
95
+ ##
96
+
97
+
98
+ @mic.eval(name="classify", dataset=tickets, output=Classification, scorers=[accuracy])
99
+ def classify(
100
+ ctx: mic.TaskContext[Classification, mic.JsonObject], ticket: Ticket
101
+ ) -> Classification:
102
+ from openai import OpenAI
103
+
104
+ # Create the client only when the task runs, and close it after the call.
105
+ with OpenAI(timeout=30, max_retries=0) as client:
106
+ response = client.responses.parse(
107
+ model="gpt-4.1-mini",
108
+ instructions=(
109
+ "Classify the support ticket as "
110
+ "bug (broken behavior), feature (new capability), or question (how-to)."
111
+ ),
112
+ input=ticket.model_dump_json(),
113
+ text_format=Classification,
114
+ store=False,
115
+ )
116
+ if response.output_parsed is None:
117
+ raise ValueError("The model did not return a classification.")
118
+ return response.output_parsed
119
+ ```
120
+
121
+ `Ticket` gives the task typed inputs; `Classification` defines both the expected
122
+ answer and the model's structured output. The SDK derives its output schema from
123
+ that class and parses the response into it. Mic validates dataset values and task
124
+ outputs against the same types, so the scorer works with objects, not JSON parsing
125
+ or string cleanup.
126
+
127
+ A valid but wrong label scores `0`; the right label scores `1`. Invalid output or
128
+ no parsed classification (for example, a refusal) is an execution failure, not a
129
+ wrong answer. These three cases are illustrative; a useful evaluation needs a
130
+ larger, representative set of labeled tickets.
131
+
132
+ Run it from the directory containing `ticket_eval.py`, then open the report:
133
+
134
+ ```console
135
+ mic run ticket_eval:classify --output .mic/tickets-first
136
+ mic report .mic/tickets-first --open
137
+ ```
138
+
139
+ That's three classifier calls. The report shows each message, its expected label,
140
+ the actual response, and its score, alongside aggregate accuracy and execution
141
+ failures. Review low-scoring cases for wrong answers or ambiguous expected labels,
142
+ and execution failures for calls that did not produce a valid classification.
143
+
144
+ Each run saves its dataset snapshot, per-trial results, and summary as JSON/JSONL,
145
+ plus a self-contained HTML report you can open without a server or network.
146
+ Explicit output directories must be empty; omit `--output` to get a unique directory
147
+ automatically. A successfully executed run can still have poor scores—execution
148
+ success and answer quality are separate.
149
+
150
+ The same code lives in [examples/ticket_eval.py](examples/ticket_eval.py). To explore
151
+ the report without credentials or API charges, the repository also includes an
152
+ [offline classification demo](examples/triage.py) with deliberately imperfect rules;
153
+ see the [walkthrough](docs/verification.md).
154
+
155
+ ## Repeat, check, improve
156
+
157
+ ### Look for variation
158
+
159
+ Run each message five times, with at most two tasks executing concurrently:
160
+
161
+ ```console
162
+ mic run ticket_eval:classify --trials 5 --concurrency 2 --output .mic/tickets-repeat
163
+ ```
164
+
165
+ This makes 15 classifier calls. Inspect individual trials as well as the mean:
166
+ one message that fails intermittently deserves attention even if the average looks
167
+ good. Repetition gives you more observations, not proof of statistical significance.
168
+
169
+ Change the prompt or model in `classify`, run again into a fresh directory,
170
+ and review both reports against the same labeled cases. Keep the evaluation set
171
+ representative rather than tuning only to these three examples.
172
+
173
+ ### Make quality a check
174
+
175
+ Add a score requirement when you're ready to use the evaluation locally or in CI:
176
+
177
+ ```console
178
+ mic run ticket_eval:classify --trials 5 --require 'accuracy>=0.9'
179
+ ```
180
+
181
+ The command exits nonzero if execution fails or mean accuracy falls below `0.9`.
182
+ That threshold is illustrative; choose one appropriate to your dataset and the cost
183
+ of a wrong answer. A failed quality gate still leaves local results to investigate.
184
+
185
+ ### Grow the dataset without changing the task
186
+
187
+ Save the same cases as `tickets.jsonl` beside `ticket_eval.py`:
188
+
189
+ ```jsonl
190
+ {"id":"upload","input":{"subject":"PDF upload","body":"The app closes whenever I upload a PDF."},"expected":{"label":"bug"}}
191
+ {"id":"export","input":{"subject":"Invoice export","body":"Can you add an option to export invoices?"},"expected":{"label":"feature"}}
192
+ {"id":"invoice","input":{"subject":"Past invoice","body":"Where can I download last month's invoice?"},"expected":{"label":"question"}}
193
+ ```
194
+
195
+ Replace the `tickets` definition with this, adding the two imports:
196
+
197
+ ```python
198
+ from pathlib import Path
199
+
200
+ from mic.providers.files import FileHandle
201
+
202
+
203
+ @mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
204
+ def tickets() -> FileHandle:
205
+ return FileHandle(Path(__file__).with_name("tickets.jsonl"))
206
+ ```
207
+
208
+ Mic hydrates the JSON objects into `Ticket` and `Classification` instances and
209
+ rejects invalid cases before calling the model. The task, scorer, and run command
210
+ stay the same. Add cases from real failures as
211
+ you encounter them—for example, a message that sounds like a feature request but
212
+ describes an existing feature that stopped working.
213
+
214
+ ## Bring your own application
215
+
216
+ The same pattern applies beyond classification: score extracted fields, check an
217
+ agent's result against a rubric, or compute a metric for an ML prediction. Tasks
218
+ and scorers are Python functions, so they can call your existing code. Sync and
219
+ async functions are supported; scripts can use `mic.run()` and notebooks can use
220
+ `await mic.arun()` instead of the CLI.
221
+
222
+ - **Structured data:** use ordinary dataclasses for inputs, outputs, and expected
223
+ values; see the [structured example](examples/structured.py) and [schema guide](docs/schemas.md).
224
+ - **Other dataset sources:** load local files, BigQuery queries, or versioned
225
+ Braintrust datasets with optional integrations; see [providers](docs/providers.md).
226
+ - **Remote reporting:** optionally export results to a Braintrust experiment after
227
+ saving local evidence. Dataset storage and reporting are independent; see [reporting](docs/reporting.md).
228
+
229
+ ## Before you use it
230
+
231
+ Mic is an early release, and its public API may change before a stable release.
232
+ It runs finite, bounded datasets locally—not distributed jobs or an application
233
+ hosting service. Synchronous callbacks must finish cooperatively: a timeout cannot
234
+ forcibly stop a running Python thread.
235
+
236
+ Reports and artifacts contain your actual evaluation data, including inputs and
237
+ outputs. They are **not automatically redacted**. Treat them as sensitive, keep
238
+ credentials out of your cases, and review evidence before sharing or committing it.
239
+ Your model calls and optional cloud integrations have their own costs and data-handling
240
+ policies. See [artifact handling](docs/artifacts.md) for details.
241
+
242
+ [API reference](docs/api.md) · [Verification](docs/verification.md) ·
243
+ [Contributing](CONTRIBUTING.md) · [Releasing](docs/releasing.md) · [MIT license](LICENSE)
@@ -0,0 +1,125 @@
1
+ # Public API and execution contract
2
+
3
+ The top-level `mic` package contains ordinary authoring, execution, schema, result, and
4
+ error APIs. Advanced provider contracts are imported from `mic.providers`; reporter
5
+ contracts are imported from `mic.reporters`. Runtime materialization and artifact
6
+ implementation modules are private. The package contains `py.typed`, and the entire source
7
+ passes strict Pyright.
8
+
9
+ ## Authoring
10
+
11
+ - `@dataset(name=..., schema=..., map_row=...)` binds a no-argument source factory.
12
+ It can return a fresh synchronous/async iterable, an awaitable of either, or a
13
+ provider handle. The factory is invoked once per run or inspection.
14
+ - `case_schema(input=..., expected=..., metadata=..., expected_policy=...)` accepts
15
+ ordinary Python annotations or `Schema[T]` adapters. Metadata defaults to `JsonObject`.
16
+ The dependency-free backend hydrates nested dataclasses from JSON mappings and
17
+ revalidates/clones native instances. Maps require string keys; native schemas are
18
+ strict and reject `strict=False`. Use explicit mapping for normalization.
19
+ Optional Pydantic adapters preserve constraints and aliases through strict JSON
20
+ hydration; canonical snapshots use Python field names. See [schemas](schemas.md).
21
+ - `@scorer(name=..., requires_expected=True)` accepts a synchronous or async
22
+ function taking `ScoreContext[I,O,E,M]`. Each scorer defines exactly one metric,
23
+ named by the scorer. Return a finite `float`/`int`, `None` when inapplicable, or
24
+ `Score(value, metadata)` when the score needs JSON metadata.
25
+ - `@eval(name=..., dataset=..., output=..., scorers=[...], trials=1, concurrency=10)`
26
+ accepts a task `(TaskContext[E,M], input: I) -> O | TaskResult[O]`, sync or async.
27
+ `TaskResult` is the only metadata wrapper. Ordinary dictionaries containing
28
+ `output` and `metadata` keys are not unpacked.
29
+ - Descriptors expose their stable `name`; callback storage is an implementation detail.
30
+ Their constructors/decorators do not call provider SDKs or authenticate. Python module
31
+ import is still ordinary code execution, not a sandbox.
32
+
33
+ ## Presence, metadata and stable identity
34
+
35
+ Input must be present. Expected must be present by default; nullable schemas may
36
+ accept `None`. Set `expected_policy="optional"` for unlabeled data. A required
37
+ scorer rejects missing labels during preflight. `require_expected()` either returns
38
+ the typed expected value or raises `MissingExpectedError`. `Missing` is encoded as
39
+ an omitted field, while `None` is JSON null.
40
+
41
+ Absent/null metadata becomes absent. Supplied metadata must validate and serialize
42
+ to an object. Task metadata is shallow-merged onto the JSON projection of dataset
43
+ metadata with task keys winning; the result is hydrated through the metadata type
44
+ again. The artifact retains dataset metadata and task metadata separately.
45
+
46
+ IDs come from mapped IDs or provider record identity. When none exists, the fallback
47
+ combines normalized row content and its position. Explicit IDs must be unique after
48
+ normalization. Physical source information remains in provenance and is excluded
49
+ from the logical dataset digest. Row ordering is part of that digest.
50
+
51
+ Serializers must preserve a meaningful JSON round trip. Nonfinite numbers fail;
52
+ there is no automatic `repr`, pickle, or silent null substitution. Validation may
53
+ execute user validators, so validators and serializers should be pure. Dataset
54
+ validation runs once before trials; output and merged metadata are validated at
55
+ their own boundaries.
56
+
57
+ ## Execution and errors
58
+
59
+ Blocking entrypoints use plain names: `run`, `preflight`, and `inspect_dataset`.
60
+ Async hosts use `await arun`, `await apreflight`, and `await ainspect_dataset`.
61
+ The run pair returns `RunResult(manifest, cases, output_dir, exit_code)`. Setup
62
+ failures raise typed `ConfigurationError`/`DatasetError`; when error artifacts are
63
+ successfully saved, the exception has a note pointing to its report. A dataset-snapshot
64
+ write failure returns exit 1 and saves an artifact-phase failure where the remaining
65
+ evidence files are writable. This also applies to filesystem initialization and
66
+ terminal manifest/report failures: computed cases remain in the returned result,
67
+ and persisted files may be incomplete or stale. Configuration/dataset errors and
68
+ cancellation retain their original exception when error persistence also fails;
69
+ secondary failures appear in exception notes. An existing nonempty output directory
70
+ is rejected without writing into it. Uninspectable schema adapters and non-callable scorers
71
+ fail before source access. Library code does not set process exit status. The CLI
72
+ translates results/errors into exit codes.
73
+
74
+ Configuration precedence is invocation arguments, decorated defaults, then library
75
+ defaults. There is no implicit `.env` loading.
76
+
77
+ All selected rows are read/validated before tasks start. `ReadLimits` defaults to
78
+ 10,000 rows, 64 MiB serialized bytes, 1 MiB per record, and a 60-second source deadline.
79
+ The default maximum is 50,000 row/trial executions. Prefix inspection is an explicit
80
+ selection; caps fail instead of truncating. These are serialized-data limits, not
81
+ an exact heap quota, and output artifact size is not currently capped separately.
82
+
83
+ Work is admitted lazily to bounded workers. Every trial gets a fresh nested copy;
84
+ every scorer gets an independent copy of pristine input/expected and validated
85
+ output/merged metadata. Scorers run in declaration order. Final API/report ordering
86
+ is row index then one-based trial number; the on-disk case journal records completion
87
+ order. Earlier successful scores remain if a later scorer fails, and the case still
88
+ fails.
89
+
90
+ Async functions are awaited; synchronous functions run in a dedicated bounded
91
+ thread executor. Trial timeout covers task and scorers. On timeout/cancellation,
92
+ running sync callbacks are drained before releasing their slots; a hanging thread
93
+ cannot be forcibly stopped. Use async callbacks and SDK request deadlines when prompt
94
+ cancellation matters. Source reads and exports also clean up cooperatively.
95
+ Cancellation preserves partial evidence then re-raises `CancelledError`; the CLI
96
+ returns 130. No tasks or scorers are retried automatically.
97
+
98
+ ## Reports and statistics
99
+
100
+ Every run writes `dataset.jsonl`, `cases.jsonl`, `run.json`, and `report.html`.
101
+ The manifest records input/expected/metadata and output schemas, selection, logical digest, source provenance, options,
102
+ source-module/framework hashes, versions, metrics, failures, and export status.
103
+ It intentionally omits environment dumps. Full evaluated values remain available.
104
+ Code provenance's framework hash covers Python sources throughout the `mic` package.
105
+ The [artifact reference](artifacts.md) specifies field shapes, partial states, and
106
+ machine-readable schemas for the current format.
107
+ Reports embed case data, and exception text and provenance may be sensitive. Read
108
+ [sensitive evidence and persistence failures](reporting.md#sensitive-evidence-and-sharing)
109
+ before sharing reports or enabling remote export.
110
+
111
+ Scores are finite numbers or `None`; booleans are rejected. Numeric means exclude
112
+ null/unavailable values. Percentiles use TypeScript's nearest-rank convention.
113
+ Numeric, null and unavailable counts are reported for every scorer.
114
+ The CLI gate grammar is `metric >= number` (also `<=`, `==`, `>`, `<`); the argument
115
+ must be shell-quoted. Missing numeric values make a configured gate unevaluable
116
+ and therefore failing. Any execution error fails regardless of the numeric mean.
117
+
118
+ Optional reporters implement `name`, `async prepare()`, and
119
+ `async report(manifest, cases) -> JsonObject`. Preparation validates configuration;
120
+ reporting runs only after complete local artifacts exist. Reporters receive copies
121
+ so they cannot mutate local evidence. Export failure/cancellation is recorded
122
+ separately and never causes task replay.
123
+ Artifact failures prevent export; failure to save a reporter's outcome prevents
124
+ subsequent reporters from starting. The completed remote write is not undone or
125
+ repeated. In-memory results retain the known outcome when the filesystem cannot.