margin-meter 0.6.0__tar.gz → 0.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {margin_meter-0.6.0 → margin_meter-0.6.2}/PKG-INFO +75 -16
- {margin_meter-0.6.0 → margin_meter-0.6.2}/README.md +74 -15
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/__init__.py +5 -1
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/__main__.py +39 -1
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/braintrust.py +11 -17
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/client.py +382 -29
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/deepeval.py +6 -2
- margin_meter-0.6.2/margin_meter/evals.py +218 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/instrument.py +141 -48
- margin_meter-0.6.2/margin_meter/langfuse.py +923 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/langsmith.py +10 -17
- margin_meter-0.6.2/margin_meter/model_override.py +279 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/promptfoo.py +15 -5
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/providers.py +166 -18
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/pytest_plugin.py +52 -6
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/ragas.py +11 -13
- margin_meter-0.6.2/margin_meter/values.py +113 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter.egg-info/PKG-INFO +75 -16
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter.egg-info/SOURCES.txt +4 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/batching.py +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/doctor.py +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter/substrate.py +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter.egg-info/dependency_links.txt +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter.egg-info/entry_points.txt +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/margin_meter.egg-info/top_level.txt +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/pyproject.toml +0 -0
- {margin_meter-0.6.0 → margin_meter-0.6.2}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: margin-meter
|
|
3
|
-
Version: 0.6.
|
|
3
|
+
Version: 0.6.2
|
|
4
4
|
Summary: Tiny stdlib-only client that auto-instruments OpenAI/Anthropic/Gemini and emits LLM call + outcome economics to a Margin ingest API.
|
|
5
5
|
Author: Margin
|
|
6
6
|
License: MIT
|
|
@@ -12,6 +12,12 @@ Description-Content-Type: text/markdown
|
|
|
12
12
|
|
|
13
13
|
# margin-meter (Python SDK)
|
|
14
14
|
|
|
15
|
+
> **You probably do not need this to start.** Margin's first step installs nothing: sign in at
|
|
16
|
+
> [trymargin.io/console](https://trymargin.io/console), connect the repository your agents live
|
|
17
|
+
> in (or drop the folder), and press Execute. Each agent runs on the evals beside it and every
|
|
18
|
+
> call is metered. This SDK is the **optional** route for metering your live production traffic,
|
|
19
|
+
> once you want that as well ([`/start#production`](https://trymargin.io/start#production)).
|
|
20
|
+
|
|
15
21
|
The tiny client a **Python** project imports to connect to Margin. It wraps your
|
|
16
22
|
LLM calls and records their outcomes, emitting each one **over HTTP** to a
|
|
17
23
|
Margin **ingest API** (`POST /api/ingest/calls` / `/api/ingest/outcomes`),
|
|
@@ -48,19 +54,30 @@ MAR-485.
|
|
|
48
54
|
Two environment variables — the deployed API base and your project's key:
|
|
49
55
|
|
|
50
56
|
```bash
|
|
51
|
-
export MARGIN_INGEST_URL="https://
|
|
52
|
-
export MARGIN_INGEST_KEY="mgk_…" #
|
|
57
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
58
|
+
export MARGIN_INGEST_KEY="mgk_…" # mint your own at https://trymargin.io/start
|
|
53
59
|
```
|
|
54
60
|
|
|
55
|
-
|
|
56
|
-
Margin repo:
|
|
61
|
+
## Get your key
|
|
57
62
|
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
63
|
+
Open **https://trymargin.io/start**, enter the email you want this project attached to, and
|
|
64
|
+
the page shows you an `mgk_…` key. It takes about ten seconds and needs nobody's approval.
|
|
65
|
+
|
|
66
|
+
The raw key is shown **once** — only its hash is stored. Keep it as `MARGIN_INGEST_KEY`.
|
|
61
67
|
|
|
62
|
-
|
|
63
|
-
project
|
|
68
|
+
⭐ **Sign in at https://console.trymargin.io with that same email** to read your own
|
|
69
|
+
cost-per-outcome. One email is one project, and the sign-in resolves through that address, so
|
|
70
|
+
the console reads exactly the rows your key writes.
|
|
71
|
+
|
|
72
|
+
⛔⛔ **THIS SECTION USED TO DESCRIBE A DIFFERENT, IMPOSSIBLE PATH, AND IT IS WORTH KNOWING
|
|
73
|
+
WHY IT SURVIVED.** It said the key was *"issued by the Margin owner"* via
|
|
74
|
+
`python3 scripts/issue_ingest_key.py <your-project-slug>` — a script inside the Margin
|
|
75
|
+
repository, which is **private**. So the documented way to get started handed a reader a
|
|
76
|
+
command they could not run, against code they could not read, on a path that needed a human
|
|
77
|
+
to be awake. Measured 2026-09-21: the self-serve route had existed the whole time and this
|
|
78
|
+
file never mentioned it once. The lesson is not "the README was stale" — it is that an
|
|
79
|
+
instruction a reader cannot execute does not fail loudly. It reads as onboarding, and the
|
|
80
|
+
reader concludes the product is not for them yet.
|
|
64
81
|
|
|
65
82
|
## Use
|
|
66
83
|
|
|
@@ -150,7 +167,7 @@ not have to hand-write a `record_outcome` call — run pytest with a workflow id
|
|
|
150
167
|
and every test becomes an outcome:
|
|
151
168
|
|
|
152
169
|
```bash
|
|
153
|
-
export MARGIN_INGEST_URL="https://
|
|
170
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
154
171
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
155
172
|
pytest --margin-workflow my-eval-suite
|
|
156
173
|
```
|
|
@@ -185,7 +202,7 @@ If your evals run under [promptfoo](https://promptfoo.dev), the JSON it already
|
|
|
185
202
|
writes is an outcome file. Run your eval, then hand the output to `margin-meter`:
|
|
186
203
|
|
|
187
204
|
```bash
|
|
188
|
-
export MARGIN_INGEST_URL="https://
|
|
205
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
189
206
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
190
207
|
promptfoo eval -o results.json
|
|
191
208
|
margin-meter promptfoo --workflow my-eval-suite results.json
|
|
@@ -224,7 +241,7 @@ Add `--simulated` for a trial run whose rows must not seat real metrics.
|
|
|
224
241
|
`$DEEPEVAL_RESULTS_FOLDER`). Point `margin-meter` at that file:
|
|
225
242
|
|
|
226
243
|
```bash
|
|
227
|
-
export MARGIN_INGEST_URL="https://
|
|
244
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
228
245
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
229
246
|
deepeval test run test_eval.py
|
|
230
247
|
margin-meter deepeval --workflow my-eval-suite .deepeval/.latest_run_full.json
|
|
@@ -247,7 +264,7 @@ per-case results to JSON (a list of results, or the `{results: […]}` /
|
|
|
247
264
|
`margin-meter` at that file:
|
|
248
265
|
|
|
249
266
|
```bash
|
|
250
|
-
export MARGIN_INGEST_URL="https://
|
|
267
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
251
268
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
252
269
|
margin-meter braintrust --workflow my-eval-suite --pass-threshold 0.7 results.json
|
|
253
270
|
```
|
|
@@ -272,7 +289,7 @@ If you evaluate RAG with [Ragas](https://docs.ragas.io), serialize the result an
|
|
|
272
289
|
point `margin-meter` at it:
|
|
273
290
|
|
|
274
291
|
```bash
|
|
275
|
-
export MARGIN_INGEST_URL="https://
|
|
292
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
276
293
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
277
294
|
python -c "import json; json.dump(result.to_pandas().to_dict(orient='records'), open('ragas.json','w'))"
|
|
278
295
|
margin-meter ragas --workflow my-rag --pass-threshold 0.7 ragas.json
|
|
@@ -296,7 +313,7 @@ If you evaluate with [LangSmith](https://docs.smith.langchain.com), serialize th
|
|
|
296
313
|
`evaluate()` result and point `margin-meter` at it:
|
|
297
314
|
|
|
298
315
|
```bash
|
|
299
|
-
export MARGIN_INGEST_URL="https://
|
|
316
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
300
317
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
301
318
|
python -c "import json; json.dump(result.to_pandas().to_dict(orient='records'), open('ls.json','w'))"
|
|
302
319
|
margin-meter langsmith --workflow my-app --pass-threshold 0.7 ls.json
|
|
@@ -315,6 +332,48 @@ evaluator keys** — LangChain's built-in judges (`correctness`, `criteria`,
|
|
|
315
332
|
records `unknown` rather than an assumed grade. Put `margin_value` on a row to
|
|
316
333
|
measure value-per-outcome; `--simulated` marks a trial run.
|
|
317
334
|
|
|
335
|
+
## Report BOTH plugs from Langfuse
|
|
336
|
+
|
|
337
|
+
[Langfuse](https://langfuse.com) is the one source that carries both halves of the two-plug
|
|
338
|
+
shape in a single export, which is why it gets two commands rather than one.
|
|
339
|
+
|
|
340
|
+
```bash
|
|
341
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
342
|
+
export MARGIN_INGEST_KEY="mgk_…"
|
|
343
|
+
|
|
344
|
+
# 1) Your AI in: every generation observation becomes a metered call.
|
|
345
|
+
margin-meter langfuse-calls --workflow my-app observations.json
|
|
346
|
+
|
|
347
|
+
# 2) Your evals in: every score becomes an outcome candidate.
|
|
348
|
+
margin-meter langfuse --workflow my-app --pass-threshold 0.7 --min-confidence 0.7 scores.json
|
|
349
|
+
```
|
|
350
|
+
|
|
351
|
+
⚠️ **A numeric score needs `--pass-threshold`; a categorical needs `--pass-values A,B`; only
|
|
352
|
+
a boolean maps to a verdict on its own.** Run `margin-meter langfuse --help` for the current
|
|
353
|
+
set — a score that arrives without the flag its data type needs is an outcome with no verdict,
|
|
354
|
+
which will not move your denominator.
|
|
355
|
+
|
|
356
|
+
Serialise either the SDK's `fetch_observations` / `fetch_scores` result or a UI export to
|
|
357
|
+
JSON. A bare list works, as do `{"data": […]}` / `{"scores": […]}` / `{"observations": […]}`
|
|
358
|
+
envelopes. No live Langfuse API key is needed here, the same as every other adapter.
|
|
359
|
+
|
|
360
|
+
⭐ **`langfuse-calls` is worth trying before you instrument anything.** A generation
|
|
361
|
+
observation already carries `model`, token usage, cost, and the trace and parent ids, so the
|
|
362
|
+
estate tree falls out of your trace hierarchy without a line of code in your agent. If your
|
|
363
|
+
generations are in Langfuse, this is the shortest path to a real cost-per-outcome.
|
|
364
|
+
|
|
365
|
+
⛔ **`--min-confidence` IS A REFUSAL, NOT A FILTER, AND THE DIFFERENCE MATTERS.** A score
|
|
366
|
+
whose `metadata` or `comment` embeds a probability is counted only at or above the threshold;
|
|
367
|
+
below it the row is **not an outcome at all** — neither a pass nor a fail. Forcing a
|
|
368
|
+
low-confidence score into a binary "least wrong answer" would fabricate the denominator that
|
|
369
|
+
cost-per-outcome divides by. A score with no stated probability is unaffected: the floor
|
|
370
|
+
gates a confidence somebody stated, it never invents one.
|
|
371
|
+
|
|
372
|
+
⚠️ **`substrate` is left empty, deliberately.** `substrate` names the ROUTE a call took
|
|
373
|
+
(direct, openrouter, bedrock), which is a different thing from the model, and a Langfuse
|
|
374
|
+
export carries no route evidence. `provider` is derived from the model string, overridable
|
|
375
|
+
per batch with `--provider`, and left `unknown` rather than fabricated when neither names it.
|
|
376
|
+
|
|
318
377
|
## Multi-agent pipelines (name each stage)
|
|
319
378
|
|
|
320
379
|
A crew of agents — a CrewAI pipeline, a planner→coder hand-off — runs in one
|
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# margin-meter (Python SDK)
|
|
2
2
|
|
|
3
|
+
> **You probably do not need this to start.** Margin's first step installs nothing: sign in at
|
|
4
|
+
> [trymargin.io/console](https://trymargin.io/console), connect the repository your agents live
|
|
5
|
+
> in (or drop the folder), and press Execute. Each agent runs on the evals beside it and every
|
|
6
|
+
> call is metered. This SDK is the **optional** route for metering your live production traffic,
|
|
7
|
+
> once you want that as well ([`/start#production`](https://trymargin.io/start#production)).
|
|
8
|
+
|
|
3
9
|
The tiny client a **Python** project imports to connect to Margin. It wraps your
|
|
4
10
|
LLM calls and records their outcomes, emitting each one **over HTTP** to a
|
|
5
11
|
Margin **ingest API** (`POST /api/ingest/calls` / `/api/ingest/outcomes`),
|
|
@@ -36,19 +42,30 @@ MAR-485.
|
|
|
36
42
|
Two environment variables — the deployed API base and your project's key:
|
|
37
43
|
|
|
38
44
|
```bash
|
|
39
|
-
export MARGIN_INGEST_URL="https://
|
|
40
|
-
export MARGIN_INGEST_KEY="mgk_…" #
|
|
45
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
46
|
+
export MARGIN_INGEST_KEY="mgk_…" # mint your own at https://trymargin.io/start
|
|
41
47
|
```
|
|
42
48
|
|
|
43
|
-
|
|
44
|
-
Margin repo:
|
|
49
|
+
## Get your key
|
|
45
50
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
51
|
+
Open **https://trymargin.io/start**, enter the email you want this project attached to, and
|
|
52
|
+
the page shows you an `mgk_…` key. It takes about ten seconds and needs nobody's approval.
|
|
53
|
+
|
|
54
|
+
The raw key is shown **once** — only its hash is stored. Keep it as `MARGIN_INGEST_KEY`.
|
|
49
55
|
|
|
50
|
-
|
|
51
|
-
project
|
|
56
|
+
⭐ **Sign in at https://console.trymargin.io with that same email** to read your own
|
|
57
|
+
cost-per-outcome. One email is one project, and the sign-in resolves through that address, so
|
|
58
|
+
the console reads exactly the rows your key writes.
|
|
59
|
+
|
|
60
|
+
⛔⛔ **THIS SECTION USED TO DESCRIBE A DIFFERENT, IMPOSSIBLE PATH, AND IT IS WORTH KNOWING
|
|
61
|
+
WHY IT SURVIVED.** It said the key was *"issued by the Margin owner"* via
|
|
62
|
+
`python3 scripts/issue_ingest_key.py <your-project-slug>` — a script inside the Margin
|
|
63
|
+
repository, which is **private**. So the documented way to get started handed a reader a
|
|
64
|
+
command they could not run, against code they could not read, on a path that needed a human
|
|
65
|
+
to be awake. Measured 2026-09-21: the self-serve route had existed the whole time and this
|
|
66
|
+
file never mentioned it once. The lesson is not "the README was stale" — it is that an
|
|
67
|
+
instruction a reader cannot execute does not fail loudly. It reads as onboarding, and the
|
|
68
|
+
reader concludes the product is not for them yet.
|
|
52
69
|
|
|
53
70
|
## Use
|
|
54
71
|
|
|
@@ -138,7 +155,7 @@ not have to hand-write a `record_outcome` call — run pytest with a workflow id
|
|
|
138
155
|
and every test becomes an outcome:
|
|
139
156
|
|
|
140
157
|
```bash
|
|
141
|
-
export MARGIN_INGEST_URL="https://
|
|
158
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
142
159
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
143
160
|
pytest --margin-workflow my-eval-suite
|
|
144
161
|
```
|
|
@@ -173,7 +190,7 @@ If your evals run under [promptfoo](https://promptfoo.dev), the JSON it already
|
|
|
173
190
|
writes is an outcome file. Run your eval, then hand the output to `margin-meter`:
|
|
174
191
|
|
|
175
192
|
```bash
|
|
176
|
-
export MARGIN_INGEST_URL="https://
|
|
193
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
177
194
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
178
195
|
promptfoo eval -o results.json
|
|
179
196
|
margin-meter promptfoo --workflow my-eval-suite results.json
|
|
@@ -212,7 +229,7 @@ Add `--simulated` for a trial run whose rows must not seat real metrics.
|
|
|
212
229
|
`$DEEPEVAL_RESULTS_FOLDER`). Point `margin-meter` at that file:
|
|
213
230
|
|
|
214
231
|
```bash
|
|
215
|
-
export MARGIN_INGEST_URL="https://
|
|
232
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
216
233
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
217
234
|
deepeval test run test_eval.py
|
|
218
235
|
margin-meter deepeval --workflow my-eval-suite .deepeval/.latest_run_full.json
|
|
@@ -235,7 +252,7 @@ per-case results to JSON (a list of results, or the `{results: […]}` /
|
|
|
235
252
|
`margin-meter` at that file:
|
|
236
253
|
|
|
237
254
|
```bash
|
|
238
|
-
export MARGIN_INGEST_URL="https://
|
|
255
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
239
256
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
240
257
|
margin-meter braintrust --workflow my-eval-suite --pass-threshold 0.7 results.json
|
|
241
258
|
```
|
|
@@ -260,7 +277,7 @@ If you evaluate RAG with [Ragas](https://docs.ragas.io), serialize the result an
|
|
|
260
277
|
point `margin-meter` at it:
|
|
261
278
|
|
|
262
279
|
```bash
|
|
263
|
-
export MARGIN_INGEST_URL="https://
|
|
280
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
264
281
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
265
282
|
python -c "import json; json.dump(result.to_pandas().to_dict(orient='records'), open('ragas.json','w'))"
|
|
266
283
|
margin-meter ragas --workflow my-rag --pass-threshold 0.7 ragas.json
|
|
@@ -284,7 +301,7 @@ If you evaluate with [LangSmith](https://docs.smith.langchain.com), serialize th
|
|
|
284
301
|
`evaluate()` result and point `margin-meter` at it:
|
|
285
302
|
|
|
286
303
|
```bash
|
|
287
|
-
export MARGIN_INGEST_URL="https://
|
|
304
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
288
305
|
export MARGIN_INGEST_KEY="mgk_…"
|
|
289
306
|
python -c "import json; json.dump(result.to_pandas().to_dict(orient='records'), open('ls.json','w'))"
|
|
290
307
|
margin-meter langsmith --workflow my-app --pass-threshold 0.7 ls.json
|
|
@@ -303,6 +320,48 @@ evaluator keys** — LangChain's built-in judges (`correctness`, `criteria`,
|
|
|
303
320
|
records `unknown` rather than an assumed grade. Put `margin_value` on a row to
|
|
304
321
|
measure value-per-outcome; `--simulated` marks a trial run.
|
|
305
322
|
|
|
323
|
+
## Report BOTH plugs from Langfuse
|
|
324
|
+
|
|
325
|
+
[Langfuse](https://langfuse.com) is the one source that carries both halves of the two-plug
|
|
326
|
+
shape in a single export, which is why it gets two commands rather than one.
|
|
327
|
+
|
|
328
|
+
```bash
|
|
329
|
+
export MARGIN_INGEST_URL="https://trymargin.io"
|
|
330
|
+
export MARGIN_INGEST_KEY="mgk_…"
|
|
331
|
+
|
|
332
|
+
# 1) Your AI in: every generation observation becomes a metered call.
|
|
333
|
+
margin-meter langfuse-calls --workflow my-app observations.json
|
|
334
|
+
|
|
335
|
+
# 2) Your evals in: every score becomes an outcome candidate.
|
|
336
|
+
margin-meter langfuse --workflow my-app --pass-threshold 0.7 --min-confidence 0.7 scores.json
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
⚠️ **A numeric score needs `--pass-threshold`; a categorical needs `--pass-values A,B`; only
|
|
340
|
+
a boolean maps to a verdict on its own.** Run `margin-meter langfuse --help` for the current
|
|
341
|
+
set — a score that arrives without the flag its data type needs is an outcome with no verdict,
|
|
342
|
+
which will not move your denominator.
|
|
343
|
+
|
|
344
|
+
Serialise either the SDK's `fetch_observations` / `fetch_scores` result or a UI export to
|
|
345
|
+
JSON. A bare list works, as do `{"data": […]}` / `{"scores": […]}` / `{"observations": […]}`
|
|
346
|
+
envelopes. No live Langfuse API key is needed here, the same as every other adapter.
|
|
347
|
+
|
|
348
|
+
⭐ **`langfuse-calls` is worth trying before you instrument anything.** A generation
|
|
349
|
+
observation already carries `model`, token usage, cost, and the trace and parent ids, so the
|
|
350
|
+
estate tree falls out of your trace hierarchy without a line of code in your agent. If your
|
|
351
|
+
generations are in Langfuse, this is the shortest path to a real cost-per-outcome.
|
|
352
|
+
|
|
353
|
+
⛔ **`--min-confidence` IS A REFUSAL, NOT A FILTER, AND THE DIFFERENCE MATTERS.** A score
|
|
354
|
+
whose `metadata` or `comment` embeds a probability is counted only at or above the threshold;
|
|
355
|
+
below it the row is **not an outcome at all** — neither a pass nor a fail. Forcing a
|
|
356
|
+
low-confidence score into a binary "least wrong answer" would fabricate the denominator that
|
|
357
|
+
cost-per-outcome divides by. A score with no stated probability is unaffected: the floor
|
|
358
|
+
gates a confidence somebody stated, it never invents one.
|
|
359
|
+
|
|
360
|
+
⚠️ **`substrate` is left empty, deliberately.** `substrate` names the ROUTE a call took
|
|
361
|
+
(direct, openrouter, bedrock), which is a different thing from the model, and a Langfuse
|
|
362
|
+
export carries no route evidence. `provider` is derived from the model string, overridable
|
|
363
|
+
per batch with `--provider`, and left `unknown` rather than fabricated when neither names it.
|
|
364
|
+
|
|
306
365
|
## Multi-agent pipelines (name each stage)
|
|
307
366
|
|
|
308
367
|
A crew of agents — a CrewAI pipeline, a planner→coder hand-off — runs in one
|
|
@@ -18,7 +18,9 @@ Two ways to instrument:
|
|
|
18
18
|
from margin_meter.batching import BATCH_PATH
|
|
19
19
|
from margin_meter.client import (
|
|
20
20
|
CALLS_PATH,
|
|
21
|
+
ESTIMATE_PATH,
|
|
21
22
|
OUTCOMES_PATH,
|
|
23
|
+
CallEstimate,
|
|
22
24
|
DeliveryStats,
|
|
23
25
|
IngestResult,
|
|
24
26
|
MarginConfigError,
|
|
@@ -61,6 +63,8 @@ __all__ = [
|
|
|
61
63
|
"MarginMeter",
|
|
62
64
|
"IngestResult",
|
|
63
65
|
"DeliveryStats",
|
|
66
|
+
"CallEstimate",
|
|
67
|
+
"ESTIMATE_PATH",
|
|
64
68
|
"MarginMeterError",
|
|
65
69
|
"MarginConfigError",
|
|
66
70
|
"MarginIngestError",
|
|
@@ -102,4 +106,4 @@ __all__ = [
|
|
|
102
106
|
# closing the invoice-reconcile coverage gap is "ship the fix, wait for the seed's
|
|
103
107
|
# dependency refresh, re-run the reconcile", and a runtime that misreports which
|
|
104
108
|
# SDK metered a call makes that unanswerable.
|
|
105
|
-
__version__ = "0.6.
|
|
109
|
+
__version__ = "0.6.2"
|
|
@@ -10,6 +10,13 @@ Subcommands:
|
|
|
10
10
|
* ``braintrust`` — report a Braintrust eval result file as outcomes (rung 2).
|
|
11
11
|
* ``ragas`` — report a Ragas evaluation result file as outcomes (rung 2).
|
|
12
12
|
* ``langsmith`` — report a LangSmith evaluate() result file as outcomes (rung 2).
|
|
13
|
+
* ``langfuse`` — report Langfuse scores as outcomes (rung 2).
|
|
14
|
+
* ``langfuse-calls`` — report Langfuse GENERATION observations as metered calls
|
|
15
|
+
(rung 2, the AI-in plug — Langfuse is the one source that carries both).
|
|
16
|
+
* ``evals`` — draft eval criteria from the connected repo, list them, or open a PR
|
|
17
|
+
with the accepted ones (MAR-1167).
|
|
18
|
+
* ``mcp`` — the same eval actions as an MCP server on stdio.
|
|
19
|
+
* ``values`` — push what each outcome is worth from your own records (MAR-1496).
|
|
13
20
|
|
|
14
21
|
Kept as a dispatcher so the surface is stable as more adapters/diagnostics arrive.
|
|
15
22
|
"""
|
|
@@ -19,7 +26,17 @@ from __future__ import annotations
|
|
|
19
26
|
import sys
|
|
20
27
|
from typing import Sequence
|
|
21
28
|
|
|
22
|
-
from margin_meter import
|
|
29
|
+
from margin_meter import (
|
|
30
|
+
braintrust,
|
|
31
|
+
deepeval,
|
|
32
|
+
doctor,
|
|
33
|
+
evals,
|
|
34
|
+
langfuse,
|
|
35
|
+
langsmith,
|
|
36
|
+
promptfoo,
|
|
37
|
+
ragas,
|
|
38
|
+
values,
|
|
39
|
+
)
|
|
23
40
|
|
|
24
41
|
_USAGE = """margin-meter — Margin's metering SDK
|
|
25
42
|
|
|
@@ -38,6 +55,17 @@ commands:
|
|
|
38
55
|
(RAG eval; pass --pass-threshold to define a pass)
|
|
39
56
|
langsmith report a LangSmith evaluate() result file to Margin as outcomes
|
|
40
57
|
(pass --pass-threshold to define a pass on continuous scores)
|
|
58
|
+
langfuse report Langfuse scores to Margin as outcomes
|
|
59
|
+
(numeric needs --pass-threshold, categorical needs --pass-values;
|
|
60
|
+
a score below --min-confidence is not recorded)
|
|
61
|
+
langfuse-calls
|
|
62
|
+
report Langfuse GENERATION observations to Margin as metered calls
|
|
63
|
+
(cost from calculatedTotalCost, else priced from tokens)
|
|
64
|
+
evals draft | list | pr: draft eval criteria from the connected repo,
|
|
65
|
+
list them, or open a PR with the ones a person accepted
|
|
66
|
+
mcp serve the evals actions as MCP tools over stdio
|
|
67
|
+
values push what each outcome is worth from your own records
|
|
68
|
+
(event_id,value_usd CSV on stdin or a file; --source, --field)
|
|
41
69
|
|
|
42
70
|
run `margin-meter <command> --help` for a command's options.
|
|
43
71
|
"""
|
|
@@ -61,6 +89,16 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
61
89
|
return ragas.main(rest)
|
|
62
90
|
if command == "langsmith":
|
|
63
91
|
return langsmith.main(rest)
|
|
92
|
+
if command == "langfuse":
|
|
93
|
+
return langfuse.main(rest)
|
|
94
|
+
if command == "langfuse-calls":
|
|
95
|
+
return langfuse.main_calls(rest)
|
|
96
|
+
if command == "evals":
|
|
97
|
+
return evals.main(rest)
|
|
98
|
+
if command == "mcp":
|
|
99
|
+
return evals.serve(rest)
|
|
100
|
+
if command == "values":
|
|
101
|
+
return values.main(rest)
|
|
64
102
|
print(f"margin-meter: unknown command {command!r}\n", file=sys.stderr)
|
|
65
103
|
print(_USAGE, file=sys.stderr)
|
|
66
104
|
return 2
|
|
@@ -36,8 +36,8 @@ The mapping, one Braintrust result → one outcome:
|
|
|
36
36
|
the grade is an LLM judge (rung 6 provenance); omitted otherwise. A bare
|
|
37
37
|
``model`` field is NOT used — in a Braintrust record it is usually the
|
|
38
38
|
generation model under test, not the judge.
|
|
39
|
-
* ``link`` ← the case ``id`` / ``span_id``, else
|
|
40
|
-
``input
|
|
39
|
+
* ``link`` ← the case ``id`` / ``span_id``, else ``braintrust:<index>`` —
|
|
40
|
+
never its ``input`` (MAR-1548).
|
|
41
41
|
* ``value_usd`` ← opt-in via ``metadata.margin_value`` — the only route to
|
|
42
42
|
value-per-outcome (rung 8). Absent is the honest cost-per-outcome default,
|
|
43
43
|
never defaulted to 0.
|
|
@@ -101,7 +101,7 @@ _JUDGE_MODEL_KEY = "margin_judge_model"
|
|
|
101
101
|
# Mirror ``src/margin/ingest.MAX_VALUE_USD`` (kept as a literal so the SDK has no
|
|
102
102
|
# dependency on the server package). A value above this is rejected by ingest and
|
|
103
103
|
# would take the whole outcome row down with it, so the adapter drops the value
|
|
104
|
-
# and keeps the row — the same treatment a
|
|
104
|
+
# and keeps the row — the same treatment a malformed value gets.
|
|
105
105
|
_MAX_VALUE_USD = 10_000_000.0
|
|
106
106
|
|
|
107
107
|
# Model ids that name no real model — a custom/deterministic scorer leaves the
|
|
@@ -296,28 +296,22 @@ def value_from_result(result: dict) -> Optional[float]:
|
|
|
296
296
|
raise ValueError(f"{_VALUE_KEY} must be a number, got {raw!r}")
|
|
297
297
|
if not math.isfinite(val):
|
|
298
298
|
raise ValueError(f"{_VALUE_KEY} must be finite, got {raw!r}")
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
raise ValueError(f"{_VALUE_KEY} implausible (> {_MAX_VALUE_USD}), got {val!r}")
|
|
299
|
+
# A loss is a real value (MAR-1198): ingest takes any finite amount within +/-MAX.
|
|
300
|
+
if abs(val) > _MAX_VALUE_USD:
|
|
301
|
+
raise ValueError(f"{_VALUE_KEY} implausible (|value| > {_MAX_VALUE_USD}), got {val!r}")
|
|
303
302
|
return val
|
|
304
303
|
|
|
305
304
|
|
|
306
305
|
def link_from_result(result: dict, index: int) -> str:
|
|
307
|
-
"""A traceable link back to the exact eval case: its id/span id, else
|
|
308
|
-
|
|
306
|
+
"""A traceable link back to the exact eval case: its id/span id, else the row
|
|
307
|
+
index.
|
|
308
|
+
|
|
309
|
+
⛔ Never the case's ``input`` (MAR-1548): /security says no prompt reaches us.
|
|
310
|
+
"""
|
|
309
311
|
for key in ("id", "span_id", "root_span_id"):
|
|
310
312
|
val = result.get(key)
|
|
311
313
|
if isinstance(val, str) and val.strip():
|
|
312
314
|
return val.strip()[:_MAX_LINK_LEN]
|
|
313
|
-
inp = result.get("input")
|
|
314
|
-
if isinstance(inp, str) and inp.strip():
|
|
315
|
-
return inp.strip()[:_MAX_LINK_LEN]
|
|
316
|
-
if isinstance(inp, (dict, list)):
|
|
317
|
-
try:
|
|
318
|
-
return json.dumps(inp, ensure_ascii=False, sort_keys=True)[:_MAX_LINK_LEN]
|
|
319
|
-
except (TypeError, ValueError):
|
|
320
|
-
pass
|
|
321
315
|
return f"braintrust:{index}"
|
|
322
316
|
|
|
323
317
|
|