mic-evals 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mic_evals-0.1.0/.gitignore +12 -0
- mic_evals-0.1.0/.python-version +1 -0
- mic_evals-0.1.0/LICENSE +21 -0
- mic_evals-0.1.0/PKG-INFO +268 -0
- mic_evals-0.1.0/README.md +243 -0
- mic_evals-0.1.0/docs/api.md +125 -0
- mic_evals-0.1.0/docs/artifact-schemas/case-record-v2.schema.json +87 -0
- mic_evals-0.1.0/docs/artifact-schemas/common-v2.schema.json +418 -0
- mic_evals-0.1.0/docs/artifact-schemas/dataset-row-v2.schema.json +22 -0
- mic_evals-0.1.0/docs/artifact-schemas/run-v2.schema.json +332 -0
- mic_evals-0.1.0/docs/artifacts.md +199 -0
- mic_evals-0.1.0/docs/providers.md +262 -0
- mic_evals-0.1.0/docs/releasing.md +67 -0
- mic_evals-0.1.0/docs/reporting.md +187 -0
- mic_evals-0.1.0/docs/schemas.md +255 -0
- mic_evals-0.1.0/docs/verification.md +128 -0
- mic_evals-0.1.0/examples/__init__.py +1 -0
- mic_evals-0.1.0/examples/async_eval.py +34 -0
- mic_evals-0.1.0/examples/failures.py +54 -0
- mic_evals-0.1.0/examples/fixtures/structured.jsonl +2 -0
- mic_evals-0.1.0/examples/fixtures/tickets.jsonl +3 -0
- mic_evals-0.1.0/examples/fixtures/triage.jsonl +3 -0
- mic_evals-0.1.0/examples/provider_equivalence.py +69 -0
- mic_evals-0.1.0/examples/pydantic_models.py +33 -0
- mic_evals-0.1.0/examples/structured.py +116 -0
- mic_evals-0.1.0/examples/ticket_eval.py +83 -0
- mic_evals-0.1.0/examples/triage.py +53 -0
- mic_evals-0.1.0/pyproject.toml +80 -0
- mic_evals-0.1.0/scripts/check_coverage.py +48 -0
- mic_evals-0.1.0/scripts/verify.py +148 -0
- mic_evals-0.1.0/scripts/verify_packaging.py +348 -0
- mic_evals-0.1.0/scripts/verify_release.py +120 -0
- mic_evals-0.1.0/scripts/verify_typing.py +54 -0
- mic_evals-0.1.0/src/mic/__init__.py +59 -0
- mic_evals-0.1.0/src/mic/__main__.py +3 -0
- mic_evals-0.1.0/src/mic/_async.py +36 -0
- mic_evals-0.1.0/src/mic/_runtime/__init__.py +1 -0
- mic_evals-0.1.0/src/mic/_runtime/artifacts.py +177 -0
- mic_evals-0.1.0/src/mic/_runtime/batch.py +78 -0
- mic_evals-0.1.0/src/mic/_runtime/callbacks.py +51 -0
- mic_evals-0.1.0/src/mic/_runtime/case.py +111 -0
- mic_evals-0.1.0/src/mic/_runtime/contracts.py +14 -0
- mic_evals-0.1.0/src/mic/_runtime/discovery.py +57 -0
- mic_evals-0.1.0/src/mic/_runtime/engine.py +279 -0
- mic_evals-0.1.0/src/mic/_runtime/files.py +27 -0
- mic_evals-0.1.0/src/mic/_runtime/materialization.py +222 -0
- mic_evals-0.1.0/src/mic/_runtime/options.py +76 -0
- mic_evals-0.1.0/src/mic/_runtime/reporting.py +68 -0
- mic_evals-0.1.0/src/mic/_runtime/summary.py +107 -0
- mic_evals-0.1.0/src/mic/_runtime/validation.py +140 -0
- mic_evals-0.1.0/src/mic/cli.py +216 -0
- mic_evals-0.1.0/src/mic/decorators.py +130 -0
- mic_evals-0.1.0/src/mic/errors.py +17 -0
- mic_evals-0.1.0/src/mic/integrations/__init__.py +3 -0
- mic_evals-0.1.0/src/mic/integrations/pydantic.py +120 -0
- mic_evals-0.1.0/src/mic/models.py +152 -0
- mic_evals-0.1.0/src/mic/providers/__init__.py +21 -0
- mic_evals-0.1.0/src/mic/providers/_braintrust/__init__.py +1 -0
- mic_evals-0.1.0/src/mic/providers/_braintrust/transport.py +184 -0
- mic_evals-0.1.0/src/mic/providers/_io.py +98 -0
- mic_evals-0.1.0/src/mic/providers/base.py +102 -0
- mic_evals-0.1.0/src/mic/providers/bigquery.py +328 -0
- mic_evals-0.1.0/src/mic/providers/braintrust.py +248 -0
- mic_evals-0.1.0/src/mic/providers/files.py +114 -0
- mic_evals-0.1.0/src/mic/providers/memory.py +78 -0
- mic_evals-0.1.0/src/mic/py.typed +0 -0
- mic_evals-0.1.0/src/mic/reporters/__init__.py +7 -0
- mic_evals-0.1.0/src/mic/reporters/base.py +15 -0
- mic_evals-0.1.0/src/mic/reporters/braintrust.py +229 -0
- mic_evals-0.1.0/src/mic/reporters/console.py +47 -0
- mic_evals-0.1.0/src/mic/reporters/html.py +82 -0
- mic_evals-0.1.0/src/mic/reporters/templates/report.html +21 -0
- mic_evals-0.1.0/src/mic/reporters/templates/report.js +219 -0
- mic_evals-0.1.0/src/mic/runner.py +5 -0
- mic_evals-0.1.0/src/mic/schema/__init__.py +64 -0
- mic_evals-0.1.0/src/mic/schema/_compiler.py +176 -0
- mic_evals-0.1.0/src/mic/schema/_contracts.py +76 -0
- mic_evals-0.1.0/src/mic/schema/_dataclasses.py +148 -0
- mic_evals-0.1.0/src/mic/schema/_values.py +236 -0
- mic_evals-0.1.0/tests/__init__.py +1 -0
- mic_evals-0.1.0/tests/acceptance/test_cli.py +189 -0
- mic_evals-0.1.0/tests/acceptance/test_cli_errors.py +85 -0
- mic_evals-0.1.0/tests/acceptance/test_readme.py +166 -0
- mic_evals-0.1.0/tests/datasets/__init__.py +1 -0
- mic_evals-0.1.0/tests/datasets/test_identity_and_cleanup.py +115 -0
- mic_evals-0.1.0/tests/datasets/test_materialization.py +310 -0
- mic_evals-0.1.0/tests/integration/test_live_providers.py +71 -0
- mic_evals-0.1.0/tests/packaging/__init__.py +1 -0
- mic_evals-0.1.0/tests/packaging/test_core_imports.py +48 -0
- mic_evals-0.1.0/tests/packaging/test_dependencies.py +80 -0
- mic_evals-0.1.0/tests/packaging/test_public_api.py +63 -0
- mic_evals-0.1.0/tests/packaging/test_release.py +172 -0
- mic_evals-0.1.0/tests/providers/__init__.py +1 -0
- mic_evals-0.1.0/tests/providers/test_bigquery_provider.py +222 -0
- mic_evals-0.1.0/tests/providers/test_bigquery_sdk_transport.py +206 -0
- mic_evals-0.1.0/tests/providers/test_braintrust_provider.py +244 -0
- mic_evals-0.1.0/tests/providers/test_braintrust_sdk_transport.py +311 -0
- mic_evals-0.1.0/tests/providers/test_file_memory_providers.py +166 -0
- mic_evals-0.1.0/tests/providers/test_provider_equivalence.py +219 -0
- mic_evals-0.1.0/tests/reporting/__init__.py +1 -0
- mic_evals-0.1.0/tests/reporting/_fixtures.py +45 -0
- mic_evals-0.1.0/tests/reporting/js/package-lock.json +546 -0
- mic_evals-0.1.0/tests/reporting/js/package.json +8 -0
- mic_evals-0.1.0/tests/reporting/js/report.test.mjs +230 -0
- mic_evals-0.1.0/tests/reporting/test_braintrust.py +352 -0
- mic_evals-0.1.0/tests/reporting/test_html.py +149 -0
- mic_evals-0.1.0/tests/runtime/__init__.py +1 -0
- mic_evals-0.1.0/tests/runtime/artifact_contract.py +54 -0
- mic_evals-0.1.0/tests/runtime/helpers.py +33 -0
- mic_evals-0.1.0/tests/runtime/test_artifact_contract.py +264 -0
- mic_evals-0.1.0/tests/runtime/test_artifacts.py +41 -0
- mic_evals-0.1.0/tests/runtime/test_cancellation.py +252 -0
- mic_evals-0.1.0/tests/runtime/test_cases.py +202 -0
- mic_evals-0.1.0/tests/runtime/test_finalization.py +341 -0
- mic_evals-0.1.0/tests/runtime/test_options.py +148 -0
- mic_evals-0.1.0/tests/runtime/test_scheduling.py +125 -0
- mic_evals-0.1.0/tests/runtime/test_setup_failures.py +109 -0
- mic_evals-0.1.0/tests/runtime/test_statistics.py +69 -0
- mic_evals-0.1.0/tests/runtime/test_structured.py +143 -0
- mic_evals-0.1.0/tests/schema/__init__.py +1 -0
- mic_evals-0.1.0/tests/schema/_fixtures.py +40 -0
- mic_evals-0.1.0/tests/schema/test_dataclasses.py +244 -0
- mic_evals-0.1.0/tests/schema/test_pydantic.py +206 -0
- mic_evals-0.1.0/tests/schema/test_values.py +195 -0
- mic_evals-0.1.0/tests/typing/negative.py +36 -0
- mic_evals-0.1.0/tests/typing/positive.py +35 -0
- mic_evals-0.1.0/tests/typing/pyrightconfig.json +17 -0
- mic_evals-0.1.0/tests/typing/structured.py +27 -0
- mic_evals-0.1.0/tests/verification/test_coverage.py +87 -0
- mic_evals-0.1.0/uv.lock +1575 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
3.12
|
mic_evals-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ryan Eiger
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
mic_evals-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: mic-evals
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Typed, provider-independent micro-evaluations for Python
|
|
5
|
+
Project-URL: Homepage, https://github.com/rybosome/mic
|
|
6
|
+
Project-URL: Repository, https://github.com/rybosome/mic
|
|
7
|
+
Project-URL: Issues, https://github.com/rybosome/mic/issues
|
|
8
|
+
Author: Ryan Eiger
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: evals,evaluation,llm,testing
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Typing :: Typed
|
|
17
|
+
Requires-Python: >=3.12
|
|
18
|
+
Provides-Extra: bigquery
|
|
19
|
+
Requires-Dist: google-cloud-bigquery<4,>=3.30; extra == 'bigquery'
|
|
20
|
+
Provides-Extra: braintrust
|
|
21
|
+
Requires-Dist: braintrust==0.39.0; extra == 'braintrust'
|
|
22
|
+
Provides-Extra: pydantic
|
|
23
|
+
Requires-Dist: pydantic<3,>=2.11; extra == 'pydantic'
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# mic
|
|
27
|
+
|
|
28
|
+
[](https://github.com/rybosome/mic/actions/workflows/ci.yml)
|
|
29
|
+
|
|
30
|
+
**Did that prompt, model, or application change actually improve the answers?**
|
|
31
|
+
|
|
32
|
+
Mic is a small Python evaluation harness for systems whose answers can vary.
|
|
33
|
+
Give it examples, the code you want to evaluate, and a way to score the results.
|
|
34
|
+
Run the evaluation repeatedly, then inspect what happened, case by case.
|
|
35
|
+
|
|
36
|
+
Use it to test an LLM classifier, score an agent's work, or evaluate another ML
|
|
37
|
+
application. You keep your application code and choose your own models and scoring
|
|
38
|
+
logic; Mic handles dataset validation, bounded concurrent execution, repeated
|
|
39
|
+
trials, and local evidence. No hosted evaluation platform is required.
|
|
40
|
+
|
|
41
|
+
## Evaluate a support-ticket classifier
|
|
42
|
+
|
|
43
|
+
Suppose you're using an LLM to sort support messages into bugs, feature requests,
|
|
44
|
+
and questions. Before changing its prompt or model, give yourself a repeatable check.
|
|
45
|
+
|
|
46
|
+
With Python 3.12+, install Mic and the SDK used by this example:
|
|
47
|
+
|
|
48
|
+
```console
|
|
49
|
+
python -m pip install "mic-evals[pydantic]" openai
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Mic's core has no third-party runtime dependencies. This example opts into Pydantic
|
|
53
|
+
and the OpenAI SDK to share a typed output contract between Mic and the model call.
|
|
54
|
+
It uses [structured outputs](https://developers.openai.com/api/docs/guides/structured-outputs)
|
|
55
|
+
with [GPT-4.1 mini](https://developers.openai.com/api/docs/models/gpt-4.1-mini);
|
|
56
|
+
replace the classifier function with your own model or application call.
|
|
57
|
+
Set `OPENAI_API_KEY` in your environment using your usual secret-management method.
|
|
58
|
+
Running it sends the example messages to OpenAI and incurs normal API charges.
|
|
59
|
+
|
|
60
|
+
Save this complete example as `ticket_eval.py`:
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
from typing import Literal
|
|
64
|
+
|
|
65
|
+
from pydantic import BaseModel
|
|
66
|
+
|
|
67
|
+
import mic
|
|
68
|
+
|
|
69
|
+
##
|
|
70
|
+
## Define a dataset
|
|
71
|
+
##
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class Ticket(BaseModel):
|
|
75
|
+
subject: str
|
|
76
|
+
body: str
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class Classification(BaseModel):
|
|
80
|
+
label: Literal["bug", "feature", "question"]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
|
|
84
|
+
def tickets() -> list[mic.RawCase]:
|
|
85
|
+
return [
|
|
86
|
+
mic.RawCase(
|
|
87
|
+
id="upload",
|
|
88
|
+
input=Ticket(subject="PDF upload", body="The app closes whenever I upload a PDF."),
|
|
89
|
+
expected=Classification(label="bug"),
|
|
90
|
+
),
|
|
91
|
+
mic.RawCase(
|
|
92
|
+
id="export",
|
|
93
|
+
input=Ticket(
|
|
94
|
+
subject="Invoice export", body="Can you add an option to export invoices?"
|
|
95
|
+
),
|
|
96
|
+
expected=Classification(label="feature"),
|
|
97
|
+
),
|
|
98
|
+
mic.RawCase(
|
|
99
|
+
id="invoice",
|
|
100
|
+
input=Ticket(subject="Past invoice", body="Where can I download last month's invoice?"),
|
|
101
|
+
expected=Classification(label="question"),
|
|
102
|
+
),
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
##
|
|
107
|
+
## Define scoring
|
|
108
|
+
##
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
@mic.scorer(name="accuracy", requires_expected=True)
|
|
112
|
+
def accuracy(
|
|
113
|
+
ctx: mic.ScoreContext[Ticket, Classification, Classification, mic.JsonObject],
|
|
114
|
+
) -> float:
|
|
115
|
+
return float(ctx.output.label == ctx.require_expected().label)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
##
|
|
119
|
+
## Define the task
|
|
120
|
+
##
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
@mic.eval(name="classify", dataset=tickets, output=Classification, scorers=[accuracy])
|
|
124
|
+
def classify(
|
|
125
|
+
ctx: mic.TaskContext[Classification, mic.JsonObject], ticket: Ticket
|
|
126
|
+
) -> Classification:
|
|
127
|
+
from openai import OpenAI
|
|
128
|
+
|
|
129
|
+
# Create the client only when the task runs, and close it after the call.
|
|
130
|
+
with OpenAI(timeout=30, max_retries=0) as client:
|
|
131
|
+
response = client.responses.parse(
|
|
132
|
+
model="gpt-4.1-mini",
|
|
133
|
+
instructions=(
|
|
134
|
+
"Classify the support ticket as "
|
|
135
|
+
"bug (broken behavior), feature (new capability), or question (how-to)."
|
|
136
|
+
),
|
|
137
|
+
input=ticket.model_dump_json(),
|
|
138
|
+
text_format=Classification,
|
|
139
|
+
store=False,
|
|
140
|
+
)
|
|
141
|
+
if response.output_parsed is None:
|
|
142
|
+
raise ValueError("The model did not return a classification.")
|
|
143
|
+
return response.output_parsed
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
`Ticket` gives the task typed inputs; `Classification` defines both the expected
|
|
147
|
+
answer and the model's structured output. The SDK derives its output schema from
|
|
148
|
+
that class and parses the response into it. Mic validates dataset values and task
|
|
149
|
+
outputs against the same types, so the scorer works with objects, not JSON parsing
|
|
150
|
+
or string cleanup.
|
|
151
|
+
|
|
152
|
+
A valid but wrong label scores `0`; the right label scores `1`. Invalid output or
|
|
153
|
+
no parsed classification (for example, a refusal) is an execution failure, not a
|
|
154
|
+
wrong answer. These three cases are illustrative; a useful evaluation needs a
|
|
155
|
+
larger, representative set of labeled tickets.
|
|
156
|
+
|
|
157
|
+
Run it from the directory containing `ticket_eval.py`, then open the report:
|
|
158
|
+
|
|
159
|
+
```console
|
|
160
|
+
mic run ticket_eval:classify --output .mic/tickets-first
|
|
161
|
+
mic report .mic/tickets-first --open
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
That's three classifier calls. The report shows each message, its expected label,
|
|
165
|
+
the actual response, and its score, alongside aggregate accuracy and execution
|
|
166
|
+
failures. Review low-scoring cases for wrong answers or ambiguous expected labels,
|
|
167
|
+
and execution failures for calls that did not produce a valid classification.
|
|
168
|
+
|
|
169
|
+
Each run saves its dataset snapshot, per-trial results, and summary as JSON/JSONL,
|
|
170
|
+
plus a self-contained HTML report you can open without a server or network.
|
|
171
|
+
Explicit output directories must be empty; omit `--output` to get a unique directory
|
|
172
|
+
automatically. A successfully executed run can still have poor scores—execution
|
|
173
|
+
success and answer quality are separate.
|
|
174
|
+
|
|
175
|
+
The same code lives in [examples/ticket_eval.py](examples/ticket_eval.py). To explore
|
|
176
|
+
the report without credentials or API charges, the repository also includes an
|
|
177
|
+
[offline classification demo](examples/triage.py) with deliberately imperfect rules;
|
|
178
|
+
see the [walkthrough](docs/verification.md).
|
|
179
|
+
|
|
180
|
+
## Repeat, check, improve
|
|
181
|
+
|
|
182
|
+
### Look for variation
|
|
183
|
+
|
|
184
|
+
Run each message five times, with at most two tasks executing concurrently:
|
|
185
|
+
|
|
186
|
+
```console
|
|
187
|
+
mic run ticket_eval:classify --trials 5 --concurrency 2 --output .mic/tickets-repeat
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
This makes 15 classifier calls. Inspect individual trials as well as the mean:
|
|
191
|
+
one message that fails intermittently deserves attention even if the average looks
|
|
192
|
+
good. Repetition gives you more observations, not proof of statistical significance.
|
|
193
|
+
|
|
194
|
+
Change the prompt or model in `classify`, run again into a fresh directory,
|
|
195
|
+
and review both reports against the same labeled cases. Keep the evaluation set
|
|
196
|
+
representative rather than tuning only to these three examples.
|
|
197
|
+
|
|
198
|
+
### Make quality a check
|
|
199
|
+
|
|
200
|
+
Add a score requirement when you're ready to use the evaluation locally or in CI:
|
|
201
|
+
|
|
202
|
+
```console
|
|
203
|
+
mic run ticket_eval:classify --trials 5 --require 'accuracy>=0.9'
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
The command exits nonzero if execution fails or mean accuracy falls below `0.9`.
|
|
207
|
+
That threshold is illustrative; choose one appropriate to your dataset and the cost
|
|
208
|
+
of a wrong answer. A failed quality gate still leaves local results to investigate.
|
|
209
|
+
|
|
210
|
+
### Grow the dataset without changing the task
|
|
211
|
+
|
|
212
|
+
Save the same cases as `tickets.jsonl` beside `ticket_eval.py`:
|
|
213
|
+
|
|
214
|
+
```jsonl
|
|
215
|
+
{"id":"upload","input":{"subject":"PDF upload","body":"The app closes whenever I upload a PDF."},"expected":{"label":"bug"}}
|
|
216
|
+
{"id":"export","input":{"subject":"Invoice export","body":"Can you add an option to export invoices?"},"expected":{"label":"feature"}}
|
|
217
|
+
{"id":"invoice","input":{"subject":"Past invoice","body":"Where can I download last month's invoice?"},"expected":{"label":"question"}}
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
Replace the `tickets` definition with this, adding the two imports:
|
|
221
|
+
|
|
222
|
+
```python
|
|
223
|
+
from pathlib import Path
|
|
224
|
+
|
|
225
|
+
from mic.providers.files import FileHandle
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
@mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
|
|
229
|
+
def tickets() -> FileHandle:
|
|
230
|
+
return FileHandle(Path(__file__).with_name("tickets.jsonl"))
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
Mic hydrates the JSON objects into `Ticket` and `Classification` instances and
|
|
234
|
+
rejects invalid cases before calling the model. The task, scorer, and run command
|
|
235
|
+
stay the same. Add cases from real failures as
|
|
236
|
+
you encounter them—for example, a message that sounds like a feature request but
|
|
237
|
+
describes an existing feature that stopped working.
|
|
238
|
+
|
|
239
|
+
## Bring your own application
|
|
240
|
+
|
|
241
|
+
The same pattern applies beyond classification: score extracted fields, check an
|
|
242
|
+
agent's result against a rubric, or compute a metric for an ML prediction. Tasks
|
|
243
|
+
and scorers are Python functions, so they can call your existing code. Sync and
|
|
244
|
+
async functions are supported; scripts can use `mic.run()` and notebooks can use
|
|
245
|
+
`await mic.arun()` instead of the CLI.
|
|
246
|
+
|
|
247
|
+
- **Structured data:** use ordinary dataclasses for inputs, outputs, and expected
|
|
248
|
+
values; see the [structured example](examples/structured.py) and [schema guide](docs/schemas.md).
|
|
249
|
+
- **Other dataset sources:** load local files, BigQuery queries, or versioned
|
|
250
|
+
Braintrust datasets with optional integrations; see [providers](docs/providers.md).
|
|
251
|
+
- **Remote reporting:** optionally export results to a Braintrust experiment after
|
|
252
|
+
saving local evidence. Dataset storage and reporting are independent; see [reporting](docs/reporting.md).
|
|
253
|
+
|
|
254
|
+
## Before you use it
|
|
255
|
+
|
|
256
|
+
Mic is an early release, and its public API may change before a stable release.
|
|
257
|
+
It runs finite, bounded datasets locally—not distributed jobs or an application
|
|
258
|
+
hosting service. Synchronous callbacks must finish cooperatively: a timeout cannot
|
|
259
|
+
forcibly stop a running Python thread.
|
|
260
|
+
|
|
261
|
+
Reports and artifacts contain your actual evaluation data, including inputs and
|
|
262
|
+
outputs. They are **not automatically redacted**. Treat them as sensitive, keep
|
|
263
|
+
credentials out of your cases, and review evidence before sharing or committing it.
|
|
264
|
+
Your model calls and optional cloud integrations have their own costs and data-handling
|
|
265
|
+
policies. See [artifact handling](docs/artifacts.md) for details.
|
|
266
|
+
|
|
267
|
+
[API reference](docs/api.md) · [Verification](docs/verification.md) ·
|
|
268
|
+
[Contributing](CONTRIBUTING.md) · [Releasing](docs/releasing.md) · [MIT license](LICENSE)
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
# mic
|
|
2
|
+
|
|
3
|
+
[](https://github.com/rybosome/mic/actions/workflows/ci.yml)
|
|
4
|
+
|
|
5
|
+
**Did that prompt, model, or application change actually improve the answers?**
|
|
6
|
+
|
|
7
|
+
Mic is a small Python evaluation harness for systems whose answers can vary.
|
|
8
|
+
Give it examples, the code you want to evaluate, and a way to score the results.
|
|
9
|
+
Run the evaluation repeatedly, then inspect what happened, case by case.
|
|
10
|
+
|
|
11
|
+
Use it to test an LLM classifier, score an agent's work, or evaluate another ML
|
|
12
|
+
application. You keep your application code and choose your own models and scoring
|
|
13
|
+
logic; Mic handles dataset validation, bounded concurrent execution, repeated
|
|
14
|
+
trials, and local evidence. No hosted evaluation platform is required.
|
|
15
|
+
|
|
16
|
+
## Evaluate a support-ticket classifier
|
|
17
|
+
|
|
18
|
+
Suppose you're using an LLM to sort support messages into bugs, feature requests,
|
|
19
|
+
and questions. Before changing its prompt or model, give yourself a repeatable check.
|
|
20
|
+
|
|
21
|
+
With Python 3.12+, install Mic and the SDK used by this example:
|
|
22
|
+
|
|
23
|
+
```console
|
|
24
|
+
python -m pip install "mic-evals[pydantic]" openai
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
Mic's core has no third-party runtime dependencies. This example opts into Pydantic
|
|
28
|
+
and the OpenAI SDK to share a typed output contract between Mic and the model call.
|
|
29
|
+
It uses [structured outputs](https://developers.openai.com/api/docs/guides/structured-outputs)
|
|
30
|
+
with [GPT-4.1 mini](https://developers.openai.com/api/docs/models/gpt-4.1-mini);
|
|
31
|
+
replace the classifier function with your own model or application call.
|
|
32
|
+
Set `OPENAI_API_KEY` in your environment using your usual secret-management method.
|
|
33
|
+
Running it sends the example messages to OpenAI and incurs normal API charges.
|
|
34
|
+
|
|
35
|
+
Save this complete example as `ticket_eval.py`:
|
|
36
|
+
|
|
37
|
+
```python
|
|
38
|
+
from typing import Literal
|
|
39
|
+
|
|
40
|
+
from pydantic import BaseModel
|
|
41
|
+
|
|
42
|
+
import mic
|
|
43
|
+
|
|
44
|
+
##
|
|
45
|
+
## Define a dataset
|
|
46
|
+
##
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class Ticket(BaseModel):
|
|
50
|
+
subject: str
|
|
51
|
+
body: str
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class Classification(BaseModel):
|
|
55
|
+
label: Literal["bug", "feature", "question"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
|
|
59
|
+
def tickets() -> list[mic.RawCase]:
|
|
60
|
+
return [
|
|
61
|
+
mic.RawCase(
|
|
62
|
+
id="upload",
|
|
63
|
+
input=Ticket(subject="PDF upload", body="The app closes whenever I upload a PDF."),
|
|
64
|
+
expected=Classification(label="bug"),
|
|
65
|
+
),
|
|
66
|
+
mic.RawCase(
|
|
67
|
+
id="export",
|
|
68
|
+
input=Ticket(
|
|
69
|
+
subject="Invoice export", body="Can you add an option to export invoices?"
|
|
70
|
+
),
|
|
71
|
+
expected=Classification(label="feature"),
|
|
72
|
+
),
|
|
73
|
+
mic.RawCase(
|
|
74
|
+
id="invoice",
|
|
75
|
+
input=Ticket(subject="Past invoice", body="Where can I download last month's invoice?"),
|
|
76
|
+
expected=Classification(label="question"),
|
|
77
|
+
),
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
##
|
|
82
|
+
## Define scoring
|
|
83
|
+
##
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@mic.scorer(name="accuracy", requires_expected=True)
|
|
87
|
+
def accuracy(
|
|
88
|
+
ctx: mic.ScoreContext[Ticket, Classification, Classification, mic.JsonObject],
|
|
89
|
+
) -> float:
|
|
90
|
+
return float(ctx.output.label == ctx.require_expected().label)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
##
|
|
94
|
+
## Define the task
|
|
95
|
+
##
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
@mic.eval(name="classify", dataset=tickets, output=Classification, scorers=[accuracy])
|
|
99
|
+
def classify(
|
|
100
|
+
ctx: mic.TaskContext[Classification, mic.JsonObject], ticket: Ticket
|
|
101
|
+
) -> Classification:
|
|
102
|
+
from openai import OpenAI
|
|
103
|
+
|
|
104
|
+
# Create the client only when the task runs, and close it after the call.
|
|
105
|
+
with OpenAI(timeout=30, max_retries=0) as client:
|
|
106
|
+
response = client.responses.parse(
|
|
107
|
+
model="gpt-4.1-mini",
|
|
108
|
+
instructions=(
|
|
109
|
+
"Classify the support ticket as "
|
|
110
|
+
"bug (broken behavior), feature (new capability), or question (how-to)."
|
|
111
|
+
),
|
|
112
|
+
input=ticket.model_dump_json(),
|
|
113
|
+
text_format=Classification,
|
|
114
|
+
store=False,
|
|
115
|
+
)
|
|
116
|
+
if response.output_parsed is None:
|
|
117
|
+
raise ValueError("The model did not return a classification.")
|
|
118
|
+
return response.output_parsed
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
`Ticket` gives the task typed inputs; `Classification` defines both the expected
|
|
122
|
+
answer and the model's structured output. The SDK derives its output schema from
|
|
123
|
+
that class and parses the response into it. Mic validates dataset values and task
|
|
124
|
+
outputs against the same types, so the scorer works with objects, not JSON parsing
|
|
125
|
+
or string cleanup.
|
|
126
|
+
|
|
127
|
+
A valid but wrong label scores `0`; the right label scores `1`. Invalid output or
|
|
128
|
+
no parsed classification (for example, a refusal) is an execution failure, not a
|
|
129
|
+
wrong answer. These three cases are illustrative; a useful evaluation needs a
|
|
130
|
+
larger, representative set of labeled tickets.
|
|
131
|
+
|
|
132
|
+
Run it from the directory containing `ticket_eval.py`, then open the report:
|
|
133
|
+
|
|
134
|
+
```console
|
|
135
|
+
mic run ticket_eval:classify --output .mic/tickets-first
|
|
136
|
+
mic report .mic/tickets-first --open
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
That's three classifier calls. The report shows each message, its expected label,
|
|
140
|
+
the actual response, and its score, alongside aggregate accuracy and execution
|
|
141
|
+
failures. Review low-scoring cases for wrong answers or ambiguous expected labels,
|
|
142
|
+
and execution failures for calls that did not produce a valid classification.
|
|
143
|
+
|
|
144
|
+
Each run saves its dataset snapshot, per-trial results, and summary as JSON/JSONL,
|
|
145
|
+
plus a self-contained HTML report you can open without a server or network.
|
|
146
|
+
Explicit output directories must be empty; omit `--output` to get a unique directory
|
|
147
|
+
automatically. A successfully executed run can still have poor scores—execution
|
|
148
|
+
success and answer quality are separate.
|
|
149
|
+
|
|
150
|
+
The same code lives in [examples/ticket_eval.py](examples/ticket_eval.py). To explore
|
|
151
|
+
the report without credentials or API charges, the repository also includes an
|
|
152
|
+
[offline classification demo](examples/triage.py) with deliberately imperfect rules;
|
|
153
|
+
see the [walkthrough](docs/verification.md).
|
|
154
|
+
|
|
155
|
+
## Repeat, check, improve
|
|
156
|
+
|
|
157
|
+
### Look for variation
|
|
158
|
+
|
|
159
|
+
Run each message five times, with at most two tasks executing concurrently:
|
|
160
|
+
|
|
161
|
+
```console
|
|
162
|
+
mic run ticket_eval:classify --trials 5 --concurrency 2 --output .mic/tickets-repeat
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
This makes 15 classifier calls. Inspect individual trials as well as the mean:
|
|
166
|
+
one message that fails intermittently deserves attention even if the average looks
|
|
167
|
+
good. Repetition gives you more observations, not proof of statistical significance.
|
|
168
|
+
|
|
169
|
+
Change the prompt or model in `classify`, run again into a fresh directory,
|
|
170
|
+
and review both reports against the same labeled cases. Keep the evaluation set
|
|
171
|
+
representative rather than tuning only to these three examples.
|
|
172
|
+
|
|
173
|
+
### Make quality a check
|
|
174
|
+
|
|
175
|
+
Add a score requirement when you're ready to use the evaluation locally or in CI:
|
|
176
|
+
|
|
177
|
+
```console
|
|
178
|
+
mic run ticket_eval:classify --trials 5 --require 'accuracy>=0.9'
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
The command exits nonzero if execution fails or mean accuracy falls below `0.9`.
|
|
182
|
+
That threshold is illustrative; choose one appropriate to your dataset and the cost
|
|
183
|
+
of a wrong answer. A failed quality gate still leaves local results to investigate.
|
|
184
|
+
|
|
185
|
+
### Grow the dataset without changing the task
|
|
186
|
+
|
|
187
|
+
Save the same cases as `tickets.jsonl` beside `ticket_eval.py`:
|
|
188
|
+
|
|
189
|
+
```jsonl
|
|
190
|
+
{"id":"upload","input":{"subject":"PDF upload","body":"The app closes whenever I upload a PDF."},"expected":{"label":"bug"}}
|
|
191
|
+
{"id":"export","input":{"subject":"Invoice export","body":"Can you add an option to export invoices?"},"expected":{"label":"feature"}}
|
|
192
|
+
{"id":"invoice","input":{"subject":"Past invoice","body":"Where can I download last month's invoice?"},"expected":{"label":"question"}}
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
Replace the `tickets` definition with this, adding the two imports:
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from pathlib import Path
|
|
199
|
+
|
|
200
|
+
from mic.providers.files import FileHandle
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
@mic.dataset(name="tickets", schema=mic.case_schema(input=Ticket, expected=Classification))
|
|
204
|
+
def tickets() -> FileHandle:
|
|
205
|
+
return FileHandle(Path(__file__).with_name("tickets.jsonl"))
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Mic hydrates the JSON objects into `Ticket` and `Classification` instances and
|
|
209
|
+
rejects invalid cases before calling the model. The task, scorer, and run command
|
|
210
|
+
stay the same. Add cases from real failures as
|
|
211
|
+
you encounter them—for example, a message that sounds like a feature request but
|
|
212
|
+
describes an existing feature that stopped working.
|
|
213
|
+
|
|
214
|
+
## Bring your own application
|
|
215
|
+
|
|
216
|
+
The same pattern applies beyond classification: score extracted fields, check an
|
|
217
|
+
agent's result against a rubric, or compute a metric for an ML prediction. Tasks
|
|
218
|
+
and scorers are Python functions, so they can call your existing code. Sync and
|
|
219
|
+
async functions are supported; scripts can use `mic.run()` and notebooks can use
|
|
220
|
+
`await mic.arun()` instead of the CLI.
|
|
221
|
+
|
|
222
|
+
- **Structured data:** use ordinary dataclasses for inputs, outputs, and expected
|
|
223
|
+
values; see the [structured example](examples/structured.py) and [schema guide](docs/schemas.md).
|
|
224
|
+
- **Other dataset sources:** load local files, BigQuery queries, or versioned
|
|
225
|
+
Braintrust datasets with optional integrations; see [providers](docs/providers.md).
|
|
226
|
+
- **Remote reporting:** optionally export results to a Braintrust experiment after
|
|
227
|
+
saving local evidence. Dataset storage and reporting are independent; see [reporting](docs/reporting.md).
|
|
228
|
+
|
|
229
|
+
## Before you use it
|
|
230
|
+
|
|
231
|
+
Mic is an early release, and its public API may change before a stable release.
|
|
232
|
+
It runs finite, bounded datasets locally—not distributed jobs or an application
|
|
233
|
+
hosting service. Synchronous callbacks must finish cooperatively: a timeout cannot
|
|
234
|
+
forcibly stop a running Python thread.
|
|
235
|
+
|
|
236
|
+
Reports and artifacts contain your actual evaluation data, including inputs and
|
|
237
|
+
outputs. They are **not automatically redacted**. Treat them as sensitive, keep
|
|
238
|
+
credentials out of your cases, and review evidence before sharing or committing it.
|
|
239
|
+
Your model calls and optional cloud integrations have their own costs and data-handling
|
|
240
|
+
policies. See [artifact handling](docs/artifacts.md) for details.
|
|
241
|
+
|
|
242
|
+
[API reference](docs/api.md) · [Verification](docs/verification.md) ·
|
|
243
|
+
[Contributing](CONTRIBUTING.md) · [Releasing](docs/releasing.md) · [MIT license](LICENSE)
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# Public API and execution contract
|
|
2
|
+
|
|
3
|
+
The top-level `mic` package contains ordinary authoring, execution, schema, result, and
|
|
4
|
+
error APIs. Advanced provider contracts are imported from `mic.providers`; reporter
|
|
5
|
+
contracts are imported from `mic.reporters`. Runtime materialization and artifact
|
|
6
|
+
implementation modules are private. The package contains `py.typed`, and the entire source
|
|
7
|
+
passes strict Pyright.
|
|
8
|
+
|
|
9
|
+
## Authoring
|
|
10
|
+
|
|
11
|
+
- `@dataset(name=..., schema=..., map_row=...)` binds a no-argument source factory.
|
|
12
|
+
It can return a fresh synchronous/async iterable, an awaitable of either, or a
|
|
13
|
+
provider handle. The factory is invoked once per run or inspection.
|
|
14
|
+
- `case_schema(input=..., expected=..., metadata=..., expected_policy=...)` accepts
|
|
15
|
+
ordinary Python annotations or `Schema[T]` adapters. Metadata defaults to `JsonObject`.
|
|
16
|
+
The dependency-free backend hydrates nested dataclasses from JSON mappings and
|
|
17
|
+
revalidates/clones native instances. Maps require string keys; native schemas are
|
|
18
|
+
strict and reject `strict=False`. Use explicit mapping for normalization.
|
|
19
|
+
Optional Pydantic adapters preserve constraints and aliases through strict JSON
|
|
20
|
+
hydration; canonical snapshots use Python field names. See [schemas](schemas.md).
|
|
21
|
+
- `@scorer(name=..., requires_expected=True)` accepts a synchronous or async
|
|
22
|
+
function taking `ScoreContext[I,O,E,M]`. Each scorer defines exactly one metric,
|
|
23
|
+
named by the scorer. Return a finite `float`/`int`, `None` when inapplicable, or
|
|
24
|
+
`Score(value, metadata)` when the score needs JSON metadata.
|
|
25
|
+
- `@eval(name=..., dataset=..., output=..., scorers=[...], trials=1, concurrency=10)`
|
|
26
|
+
accepts a task `(TaskContext[E,M], input: I) -> O | TaskResult[O]`, sync or async.
|
|
27
|
+
`TaskResult` is the only metadata wrapper. Ordinary dictionaries containing
|
|
28
|
+
`output` and `metadata` keys are not unpacked.
|
|
29
|
+
- Descriptors expose their stable `name`; callback storage is an implementation detail.
|
|
30
|
+
Their constructors/decorators do not call provider SDKs or authenticate. Python module
|
|
31
|
+
import is still ordinary code execution, not a sandbox.
|
|
32
|
+
|
|
33
|
+
## Presence, metadata and stable identity
|
|
34
|
+
|
|
35
|
+
Input must be present. Expected must be present by default; nullable schemas may
|
|
36
|
+
accept `None`. Set `expected_policy="optional"` for unlabeled data. A required
|
|
37
|
+
scorer rejects missing labels during preflight. `require_expected()` either returns
|
|
38
|
+
the typed expected value or raises `MissingExpectedError`. `Missing` is encoded as
|
|
39
|
+
an omitted field, while `None` is JSON null.
|
|
40
|
+
|
|
41
|
+
Absent/null metadata becomes absent. Supplied metadata must validate and serialize
|
|
42
|
+
to an object. Task metadata is shallow-merged onto the JSON projection of dataset
|
|
43
|
+
metadata with task keys winning; the result is hydrated through the metadata type
|
|
44
|
+
again. The artifact retains dataset metadata and task metadata separately.
|
|
45
|
+
|
|
46
|
+
IDs come from mapped IDs or provider record identity. When none exists, the fallback
|
|
47
|
+
combines normalized row content and its position. Explicit IDs must be unique after
|
|
48
|
+
normalization. Physical source information remains in provenance and is excluded
|
|
49
|
+
from the logical dataset digest. Row ordering is part of that digest.
|
|
50
|
+
|
|
51
|
+
Serializers must preserve a meaningful JSON round trip. Nonfinite numbers fail;
|
|
52
|
+
there is no automatic `repr`, pickle, or silent null substitution. Validation may
|
|
53
|
+
execute user validators, so validators and serializers should be pure. Dataset
|
|
54
|
+
validation runs once before trials; output and merged metadata are validated at
|
|
55
|
+
their own boundaries.
|
|
56
|
+
|
|
57
|
+
## Execution and errors
|
|
58
|
+
|
|
59
|
+
Blocking entrypoints use plain names: `run`, `preflight`, and `inspect_dataset`.
|
|
60
|
+
Async hosts use `await arun`, `await apreflight`, and `await ainspect_dataset`.
|
|
61
|
+
The run pair returns `RunResult(manifest, cases, output_dir, exit_code)`. Setup
|
|
62
|
+
failures raise typed `ConfigurationError`/`DatasetError`; when error artifacts are
|
|
63
|
+
successfully saved, the exception has a note pointing to its report. A dataset-snapshot
|
|
64
|
+
write failure returns exit 1 and saves an artifact-phase failure where the remaining
|
|
65
|
+
evidence files are writable. This also applies to filesystem initialization and
|
|
66
|
+
terminal manifest/report failures: computed cases remain in the returned result,
|
|
67
|
+
and persisted files may be incomplete or stale. Configuration/dataset errors and
|
|
68
|
+
cancellation retain their original exception when error persistence also fails;
|
|
69
|
+
secondary failures appear in exception notes. An existing nonempty output directory
|
|
70
|
+
is rejected without writing into it. Uninspectable schema adapters and non-callable scorers
|
|
71
|
+
fail before source access. Library code does not set process exit status. The CLI
|
|
72
|
+
translates results/errors into exit codes.
|
|
73
|
+
|
|
74
|
+
Configuration precedence is invocation arguments, decorated defaults, then library
|
|
75
|
+
defaults. There is no implicit `.env` loading.
|
|
76
|
+
|
|
77
|
+
All selected rows are read/validated before tasks start. `ReadLimits` defaults to
|
|
78
|
+
10,000 rows, 64 MiB serialized bytes, 1 MiB per record, and a 60-second source deadline.
|
|
79
|
+
The default maximum is 50,000 row/trial executions. Prefix inspection is an explicit
|
|
80
|
+
selection; caps fail instead of truncating. These are serialized-data limits, not
|
|
81
|
+
an exact heap quota, and output artifact size is not currently capped separately.
|
|
82
|
+
|
|
83
|
+
Work is admitted lazily to bounded workers. Every trial gets a fresh nested copy;
|
|
84
|
+
every scorer gets an independent copy of pristine input/expected and validated
|
|
85
|
+
output/merged metadata. Scorers run in declaration order. Final API/report ordering
|
|
86
|
+
is row index then one-based trial number; the on-disk case journal records completion
|
|
87
|
+
order. Earlier successful scores remain if a later scorer fails, and the case still
|
|
88
|
+
fails.
|
|
89
|
+
|
|
90
|
+
Async functions are awaited; synchronous functions run in a dedicated bounded
|
|
91
|
+
thread executor. Trial timeout covers task and scorers. On timeout/cancellation,
|
|
92
|
+
running sync callbacks are drained before releasing their slots; a hanging thread
|
|
93
|
+
cannot be forcibly stopped. Use async callbacks and SDK request deadlines when prompt
|
|
94
|
+
cancellation matters. Source reads and exports also clean up cooperatively.
|
|
95
|
+
Cancellation preserves partial evidence then re-raises `CancelledError`; the CLI
|
|
96
|
+
returns 130. No tasks or scorers are retried automatically.
|
|
97
|
+
|
|
98
|
+
## Reports and statistics
|
|
99
|
+
|
|
100
|
+
Every run writes `dataset.jsonl`, `cases.jsonl`, `run.json`, and `report.html`.
|
|
101
|
+
The manifest records input/expected/metadata and output schemas, selection, logical digest, source provenance, options,
|
|
102
|
+
source-module/framework hashes, versions, metrics, failures, and export status.
|
|
103
|
+
It intentionally omits environment dumps. Full evaluated values remain available.
|
|
104
|
+
Code provenance's framework hash covers Python sources throughout the `mic` package.
|
|
105
|
+
The [artifact reference](artifacts.md) specifies field shapes, partial states, and
|
|
106
|
+
machine-readable schemas for the current format.
|
|
107
|
+
Reports embed case data, and exception text and provenance may be sensitive. Read
|
|
108
|
+
[sensitive evidence and persistence failures](reporting.md#sensitive-evidence-and-sharing)
|
|
109
|
+
before sharing reports or enabling remote export.
|
|
110
|
+
|
|
111
|
+
Scores are finite numbers or `None`; booleans are rejected. Numeric means exclude
|
|
112
|
+
null/unavailable values. Percentiles use TypeScript's nearest-rank convention.
|
|
113
|
+
Numeric, null and unavailable counts are reported for every scorer.
|
|
114
|
+
The CLI gate grammar is `metric >= number` (also `<=`, `==`, `>`, `<`); the argument
|
|
115
|
+
must be shell-quoted. Missing numeric values make a configured gate unevaluable
|
|
116
|
+
and therefore failing. Any execution error fails regardless of the numeric mean.
|
|
117
|
+
|
|
118
|
+
Optional reporters implement `name`, `async prepare()`, and
|
|
119
|
+
`async report(manifest, cases) -> JsonObject`. Preparation validates configuration;
|
|
120
|
+
reporting runs only after complete local artifacts exist. Reporters receive copies
|
|
121
|
+
so they cannot mutate local evidence. Export failure/cancellation is recorded
|
|
122
|
+
separately and never causes task replay.
|
|
123
|
+
Artifact failures prevent export; failure to save a reporter's outcome prevents
|
|
124
|
+
subsequent reporters from starting. The completed remote write is not undone or
|
|
125
|
+
repeated. In-memory results retain the known outcome when the filesystem cannot.
|