assay-evals 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- assay_evals-0.1.0/LICENSE +21 -0
- assay_evals-0.1.0/PKG-INFO +82 -0
- assay_evals-0.1.0/README.md +58 -0
- assay_evals-0.1.0/assay_evals.egg-info/PKG-INFO +82 -0
- assay_evals-0.1.0/assay_evals.egg-info/SOURCES.txt +9 -0
- assay_evals-0.1.0/assay_evals.egg-info/dependency_links.txt +1 -0
- assay_evals-0.1.0/assay_evals.egg-info/top_level.txt +1 -0
- assay_evals-0.1.0/assay_sdk/__init__.py +361 -0
- assay_evals-0.1.0/assay_sdk/py.typed +0 -0
- assay_evals-0.1.0/pyproject.toml +37 -0
- assay_evals-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tap222
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: assay-evals
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results
|
|
5
|
+
Author: tap222
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/tap222/docai-eval
|
|
8
|
+
Project-URL: Documentation, https://github.com/tap222/docai-eval/tree/main/sdk/python#readme
|
|
9
|
+
Project-URL: Event schema, https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md
|
|
10
|
+
Project-URL: Issues, https://github.com/tap222/docai-eval/issues
|
|
11
|
+
Keywords: llm,agents,evaluation,observability,tracing,ai
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Software Development :: Testing
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# assay-evals
|
|
26
|
+
|
|
27
|
+
Record what your AI system does, and how it went, in
|
|
28
|
+
[Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
|
|
29
|
+
results. Standard library only, Python 3.9+.
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install assay-evals
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
You need an Assay server to send to. See the
|
|
36
|
+
[setup guide](https://github.com/tap222/docai-eval#setup-guide).
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
import assay_sdk as assay
|
|
40
|
+
|
|
41
|
+
assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
|
|
42
|
+
|
|
43
|
+
# An agent
|
|
44
|
+
with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
|
|
45
|
+
run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
|
|
46
|
+
order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
|
|
47
|
+
run.state("refund:O-17", "create", {"amount": order["price"]})
|
|
48
|
+
run.answer(f"Refunded ${order['price']}.")
|
|
49
|
+
|
|
50
|
+
# A pipeline
|
|
51
|
+
with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
|
|
52
|
+
with run.stage("extract", prompt="extract_fields@v13") as s:
|
|
53
|
+
run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
|
|
54
|
+
s.outputs.update(fields)
|
|
55
|
+
|
|
56
|
+
# Outcomes, whenever they're known
|
|
57
|
+
assay.feedback(run.id, "thumbs_down")
|
|
58
|
+
assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
|
|
59
|
+
assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
|
|
60
|
+
assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
| Call | Records |
|
|
64
|
+
|---|---|
|
|
65
|
+
| `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
|
|
66
|
+
| `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
|
|
67
|
+
| `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
|
|
68
|
+
| `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
|
|
69
|
+
|
|
70
|
+
These options go to `init()`:
|
|
71
|
+
- `redact`: a function applied to inputs, arguments, results, text and outputs before they
|
|
72
|
+
leave the process.
|
|
73
|
+
- `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
|
|
74
|
+
are always sent.
|
|
75
|
+
- `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
|
|
76
|
+
your code.
|
|
77
|
+
- `enabled=False`: the SDK does nothing, e.g. in unit tests.
|
|
78
|
+
|
|
79
|
+
Events follow the
|
|
80
|
+
[Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
|
|
81
|
+
`POST /v1/ingest` in the background, so a run that crashes still shows every step up to
|
|
82
|
+
the crash.
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# assay-evals
|
|
2
|
+
|
|
3
|
+
Record what your AI system does, and how it went, in
|
|
4
|
+
[Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
|
|
5
|
+
results. Standard library only, Python 3.9+.
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install assay-evals
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
You need an Assay server to send to. See the
|
|
12
|
+
[setup guide](https://github.com/tap222/docai-eval#setup-guide).
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
import assay_sdk as assay
|
|
16
|
+
|
|
17
|
+
assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
|
|
18
|
+
|
|
19
|
+
# An agent
|
|
20
|
+
with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
|
|
21
|
+
run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
|
|
22
|
+
order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
|
|
23
|
+
run.state("refund:O-17", "create", {"amount": order["price"]})
|
|
24
|
+
run.answer(f"Refunded ${order['price']}.")
|
|
25
|
+
|
|
26
|
+
# A pipeline
|
|
27
|
+
with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
|
|
28
|
+
with run.stage("extract", prompt="extract_fields@v13") as s:
|
|
29
|
+
run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
|
|
30
|
+
s.outputs.update(fields)
|
|
31
|
+
|
|
32
|
+
# Outcomes, whenever they're known
|
|
33
|
+
assay.feedback(run.id, "thumbs_down")
|
|
34
|
+
assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
|
|
35
|
+
assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
|
|
36
|
+
assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
| Call | Records |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
|
|
42
|
+
| `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
|
|
43
|
+
| `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
|
|
44
|
+
| `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
|
|
45
|
+
|
|
46
|
+
These options go to `init()`:
|
|
47
|
+
- `redact`: a function applied to inputs, arguments, results, text and outputs before they
|
|
48
|
+
leave the process.
|
|
49
|
+
- `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
|
|
50
|
+
are always sent.
|
|
51
|
+
- `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
|
|
52
|
+
your code.
|
|
53
|
+
- `enabled=False`: the SDK does nothing, e.g. in unit tests.
|
|
54
|
+
|
|
55
|
+
Events follow the
|
|
56
|
+
[Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
|
|
57
|
+
`POST /v1/ingest` in the background, so a run that crashes still shows every step up to
|
|
58
|
+
the crash.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: assay-evals
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results
|
|
5
|
+
Author: tap222
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/tap222/docai-eval
|
|
8
|
+
Project-URL: Documentation, https://github.com/tap222/docai-eval/tree/main/sdk/python#readme
|
|
9
|
+
Project-URL: Event schema, https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md
|
|
10
|
+
Project-URL: Issues, https://github.com/tap222/docai-eval/issues
|
|
11
|
+
Keywords: llm,agents,evaluation,observability,tracing,ai
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Software Development :: Testing
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# assay-evals
|
|
26
|
+
|
|
27
|
+
Record what your AI system does, and how it went, in
|
|
28
|
+
[Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
|
|
29
|
+
results. Standard library only, Python 3.9+.
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install assay-evals
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
You need an Assay server to send to. See the
|
|
36
|
+
[setup guide](https://github.com/tap222/docai-eval#setup-guide).
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
import assay_sdk as assay
|
|
40
|
+
|
|
41
|
+
assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
|
|
42
|
+
|
|
43
|
+
# An agent
|
|
44
|
+
with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
|
|
45
|
+
run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
|
|
46
|
+
order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
|
|
47
|
+
run.state("refund:O-17", "create", {"amount": order["price"]})
|
|
48
|
+
run.answer(f"Refunded ${order['price']}.")
|
|
49
|
+
|
|
50
|
+
# A pipeline
|
|
51
|
+
with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
|
|
52
|
+
with run.stage("extract", prompt="extract_fields@v13") as s:
|
|
53
|
+
run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
|
|
54
|
+
s.outputs.update(fields)
|
|
55
|
+
|
|
56
|
+
# Outcomes, whenever they're known
|
|
57
|
+
assay.feedback(run.id, "thumbs_down")
|
|
58
|
+
assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
|
|
59
|
+
assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
|
|
60
|
+
assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
| Call | Records |
|
|
64
|
+
|---|---|
|
|
65
|
+
| `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
|
|
66
|
+
| `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
|
|
67
|
+
| `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
|
|
68
|
+
| `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
|
|
69
|
+
|
|
70
|
+
These options go to `init()`:
|
|
71
|
+
- `redact`: a function applied to inputs, arguments, results, text and outputs before they
|
|
72
|
+
leave the process.
|
|
73
|
+
- `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
|
|
74
|
+
are always sent.
|
|
75
|
+
- `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
|
|
76
|
+
your code.
|
|
77
|
+
- `enabled=False`: the SDK does nothing, e.g. in unit tests.
|
|
78
|
+
|
|
79
|
+
Events follow the
|
|
80
|
+
[Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
|
|
81
|
+
`POST /v1/ingest` in the background, so a run that crashes still shows every step up to
|
|
82
|
+
the crash.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
assay_sdk
|
|
@@ -0,0 +1,361 @@
|
|
|
1
|
+
"""Assay SDK: record what your AI system does, and how it went. Standard library only.
|
|
2
|
+
|
|
3
|
+
import assay_sdk as assay
|
|
4
|
+
|
|
5
|
+
assay.init("https://assay.example.com", key="ak_...") # or ASSAY_URL / ASSAY_KEY
|
|
6
|
+
|
|
7
|
+
with assay.run("refund_request", input=message, version={"prompt": "support@v5"}) as run:
|
|
8
|
+
run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
|
|
9
|
+
order = run.call("get_order", get_order, order_id="O-17") # runs it, records result or error
|
|
10
|
+
run.state("refund:O-17", "create", {"amount": 27.61})
|
|
11
|
+
run.answer("Refunded $27.61.")
|
|
12
|
+
|
|
13
|
+
assay.feedback(run.id, "thumbs_up") # later, from your UI
|
|
14
|
+
assay.check("nightly-0924", "case-17", "pass", run_id=run.id, field="answer") # from your tests
|
|
15
|
+
|
|
16
|
+
Events follow the Assay event schema v1 (docs/event-schema.md) and stream to
|
|
17
|
+
POST /v1/ingest in the background: a run that crashes still shows every step up
|
|
18
|
+
to that point. Every event has an id, so retries never duplicate anything. The
|
|
19
|
+
SDK never raises into your code (pass strict=True to init while developing).
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import atexit
|
|
24
|
+
import json
|
|
25
|
+
import logging
|
|
26
|
+
import os
|
|
27
|
+
import random
|
|
28
|
+
import threading
|
|
29
|
+
import time
|
|
30
|
+
import urllib.error
|
|
31
|
+
import urllib.request
|
|
32
|
+
import uuid
|
|
33
|
+
from contextlib import contextmanager
|
|
34
|
+
from datetime import datetime, timezone
|
|
35
|
+
from typing import Any, Callable, Dict, List, Optional
|
|
36
|
+
|
|
37
|
+
__all__ = ["init", "run", "feedback", "check", "correction", "expect", "flush", "shutdown", "Run"]
|
|
38
|
+
__version__ = "0.1.0"
|
|
39
|
+
|
|
40
|
+
log = logging.getLogger("assay_sdk")
|
|
41
|
+
SCHEMA = 1
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _now() -> str:
|
|
45
|
+
return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _ts(t: Optional[datetime]) -> Optional[str]:
|
|
49
|
+
if t is None:
|
|
50
|
+
return None
|
|
51
|
+
if t.tzinfo is None:
|
|
52
|
+
t = t.replace(tzinfo=timezone.utc)
|
|
53
|
+
return t.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _json_safe(v):
|
|
57
|
+
"""Make a value JSON-serializable without failing: unknown objects become their repr."""
|
|
58
|
+
try:
|
|
59
|
+
json.dumps(v)
|
|
60
|
+
return v
|
|
61
|
+
except (TypeError, ValueError):
|
|
62
|
+
if isinstance(v, dict):
|
|
63
|
+
return {str(k): _json_safe(x) for k, x in v.items()}
|
|
64
|
+
if isinstance(v, (list, tuple, set)):
|
|
65
|
+
return [_json_safe(x) for x in v]
|
|
66
|
+
if isinstance(v, datetime):
|
|
67
|
+
return _ts(v)
|
|
68
|
+
return repr(v)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class _Client:
|
|
72
|
+
def __init__(self, url: str, key: Optional[str], tenant: Optional[str], redact: Optional[Callable],
|
|
73
|
+
sample: float, flush_interval: float, batch_size: int, max_queue: int, strict: bool,
|
|
74
|
+
transport: Optional[Callable[[List[dict]], None]], enabled: bool):
|
|
75
|
+
self.url, self.key, self.tenant = url.rstrip("/"), key, tenant
|
|
76
|
+
self.redact, self.sample, self.strict, self.enabled = redact, sample, strict, enabled
|
|
77
|
+
self.batch_size, self.max_queue = batch_size, max_queue
|
|
78
|
+
self._send = transport or self._http
|
|
79
|
+
self._queue: List[dict] = []
|
|
80
|
+
self._lock = threading.Lock()
|
|
81
|
+
self._wake = threading.Event()
|
|
82
|
+
self._stop = False
|
|
83
|
+
self.dropped = 0
|
|
84
|
+
self._thread = threading.Thread(target=self._loop, args=(flush_interval,), name="assay-evals", daemon=True)
|
|
85
|
+
self._thread.start()
|
|
86
|
+
|
|
87
|
+
def emit(self, event: dict) -> None:
|
|
88
|
+
if not self.enabled:
|
|
89
|
+
return
|
|
90
|
+
event = {"v": SCHEMA, "id": uuid.uuid4().hex, "ts": _now(), **event}
|
|
91
|
+
with self._lock:
|
|
92
|
+
if len(self._queue) >= self.max_queue:
|
|
93
|
+
self._queue.pop(0)
|
|
94
|
+
self.dropped += 1
|
|
95
|
+
if self.dropped in (1, 100, 10000):
|
|
96
|
+
log.warning("Assay: queue full, dropped %d events so far (is the server reachable?)", self.dropped)
|
|
97
|
+
self._queue.append({k: v for k, v in event.items() if v is not None})
|
|
98
|
+
full = len(self._queue) >= self.batch_size
|
|
99
|
+
if full:
|
|
100
|
+
self._wake.set()
|
|
101
|
+
|
|
102
|
+
def clean(self, v):
|
|
103
|
+
"""Redact (if configured), then make JSON-safe."""
|
|
104
|
+
if v is None:
|
|
105
|
+
return None
|
|
106
|
+
if self.redact:
|
|
107
|
+
try:
|
|
108
|
+
v = self.redact(v)
|
|
109
|
+
except Exception:
|
|
110
|
+
log.exception("Assay: redact() failed; dropping the value rather than sending it unredacted")
|
|
111
|
+
return "<redaction failed>"
|
|
112
|
+
return _json_safe(v)
|
|
113
|
+
|
|
114
|
+
def flush(self) -> bool:
|
|
115
|
+
"""Send everything queued. True if the queue is empty afterwards."""
|
|
116
|
+
while True:
|
|
117
|
+
with self._lock:
|
|
118
|
+
batch, self._queue = self._queue[:self.batch_size], self._queue[self.batch_size:]
|
|
119
|
+
if not batch:
|
|
120
|
+
return True
|
|
121
|
+
try:
|
|
122
|
+
self._send(batch)
|
|
123
|
+
except Exception:
|
|
124
|
+
with self._lock: # keep them, in order, for the next try
|
|
125
|
+
self._queue = batch + self._queue
|
|
126
|
+
if self.strict:
|
|
127
|
+
raise
|
|
128
|
+
log.warning("Assay: couldn't send %d events; will retry", len(batch), exc_info=True)
|
|
129
|
+
return False
|
|
130
|
+
|
|
131
|
+
def _loop(self, interval: float) -> None:
|
|
132
|
+
backoff = interval
|
|
133
|
+
while not self._stop:
|
|
134
|
+
self._wake.wait(backoff)
|
|
135
|
+
self._wake.clear()
|
|
136
|
+
ok = self.flush() if not self.strict else self._flush_quiet()
|
|
137
|
+
backoff = interval if ok else min(backoff * 2, 60.0)
|
|
138
|
+
|
|
139
|
+
def _flush_quiet(self) -> bool:
|
|
140
|
+
try:
|
|
141
|
+
return self.flush()
|
|
142
|
+
except Exception:
|
|
143
|
+
return False
|
|
144
|
+
|
|
145
|
+
def close(self) -> None:
|
|
146
|
+
self._stop = True
|
|
147
|
+
self._wake.set()
|
|
148
|
+
self.flush()
|
|
149
|
+
|
|
150
|
+
def _http(self, batch: List[dict]) -> None:
|
|
151
|
+
headers = {"Content-Type": "application/json", "User-Agent": f"assay-evals-python/{__version__}"}
|
|
152
|
+
if self.key:
|
|
153
|
+
headers["Authorization"] = f"Bearer {self.key}"
|
|
154
|
+
if self.tenant:
|
|
155
|
+
headers["X-Tenant"] = self.tenant
|
|
156
|
+
req = urllib.request.Request(self.url + "/v1/ingest", data=json.dumps({"events": batch}).encode(),
|
|
157
|
+
headers=headers, method="POST")
|
|
158
|
+
for attempt in range(3):
|
|
159
|
+
try:
|
|
160
|
+
with urllib.request.urlopen(req, timeout=10) as r:
|
|
161
|
+
r.read()
|
|
162
|
+
return
|
|
163
|
+
except urllib.error.HTTPError as e:
|
|
164
|
+
if e.code == 422: # a malformed event won't improve with retries: log it and drop the batch
|
|
165
|
+
log.error("Assay rejected events: %s", e.read().decode()[:1000])
|
|
166
|
+
return
|
|
167
|
+
if e.code < 500 and e.code != 429 or attempt == 2:
|
|
168
|
+
raise
|
|
169
|
+
except OSError:
|
|
170
|
+
if attempt == 2:
|
|
171
|
+
raise
|
|
172
|
+
time.sleep(0.5 * 2 ** attempt)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
_client: Optional[_Client] = None
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def init(url: Optional[str] = None, key: Optional[str] = None, *, tenant: Optional[str] = None,
|
|
179
|
+
redact: Optional[Callable[[Any], Any]] = None, sample: float = 1.0, flush_interval: float = 1.0,
|
|
180
|
+
batch_size: int = 500, max_queue: int = 100_000, strict: bool = False,
|
|
181
|
+
transport: Optional[Callable[[List[dict]], None]] = None, enabled: bool = True) -> None:
|
|
182
|
+
"""Configure the SDK once, at startup.
|
|
183
|
+
|
|
184
|
+
url / key default to the ASSAY_URL / ASSAY_KEY environment variables
|
|
185
|
+
redact applied to inputs, arguments, results, text and outputs before they leave the process
|
|
186
|
+
sample share of runs to record (0.1 = one in ten); outcomes are always sent
|
|
187
|
+
strict raise send errors instead of logging them (for development)
|
|
188
|
+
enabled False turns the SDK into a no-op (e.g. in unit tests)
|
|
189
|
+
"""
|
|
190
|
+
global _client
|
|
191
|
+
if _client is not None:
|
|
192
|
+
_client.close()
|
|
193
|
+
url = url or os.environ.get("ASSAY_URL")
|
|
194
|
+
if not url and transport is None and enabled:
|
|
195
|
+
raise ValueError("Give init() a url, or set ASSAY_URL.")
|
|
196
|
+
_client = _Client(url or "http://localhost", key or os.environ.get("ASSAY_KEY"), tenant, redact, sample,
|
|
197
|
+
flush_interval, batch_size, max_queue, strict, transport, enabled)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _c() -> _Client:
|
|
201
|
+
if _client is None:
|
|
202
|
+
init()
|
|
203
|
+
return _client
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
class Run:
|
|
207
|
+
"""One run of your system on one input. Use through `assay.run(...)`."""
|
|
208
|
+
|
|
209
|
+
def __init__(self, client: _Client, run_id: str, recorded: bool):
|
|
210
|
+
self.id, self._c, self._on = run_id, client, recorded
|
|
211
|
+
self._seq, self._lock, self._parents = 0, threading.Lock(), []
|
|
212
|
+
self.answer_text: Optional[str] = None
|
|
213
|
+
|
|
214
|
+
def _step(self, kind: str, started: Optional[datetime] = None, **fields) -> int:
|
|
215
|
+
with self._lock:
|
|
216
|
+
seq, self._seq = self._seq, self._seq + 1
|
|
217
|
+
parent = self._parents[-1] if self._parents else None
|
|
218
|
+
if self._on:
|
|
219
|
+
self._c.emit({"type": "step", "run_id": self.id, "seq": seq, "kind": kind, "parent_seq": parent,
|
|
220
|
+
"ts": _ts(started) or _now(), **fields})
|
|
221
|
+
return seq
|
|
222
|
+
|
|
223
|
+
def llm(self, model: Optional[str] = None, tokens_in: Optional[int] = None, tokens_out: Optional[int] = None,
|
|
224
|
+
cost_usd: Optional[float] = None, prompt: Optional[str] = None, text: Optional[str] = None,
|
|
225
|
+
started: Optional[datetime] = None, ended: Optional[datetime] = None, error: Optional[str] = None) -> None:
|
|
226
|
+
"""A model call. prompt is "id@version"; text is the output (or a summary of it)."""
|
|
227
|
+
self._step("llm", started, model=model, tokens_in=tokens_in, tokens_out=tokens_out, cost_usd=cost_usd,
|
|
228
|
+
prompt=prompt, text=self._c.clean(text), ended_at=_ts(ended),
|
|
229
|
+
status="error" if error else "ok", error=error)
|
|
230
|
+
|
|
231
|
+
def tool(self, name: str, args: Optional[Dict[str, Any]] = None, result: Any = None, error: Optional[str] = None,
|
|
232
|
+
started: Optional[datetime] = None, ended: Optional[datetime] = None) -> None:
|
|
233
|
+
"""A tool call you've already made."""
|
|
234
|
+
self._step("tool", started, name=name, args=self._c.clean(args or {}), result=self._c.clean(result),
|
|
235
|
+
ended_at=_ts(ended), status="error" if error else "ok", error=error)
|
|
236
|
+
|
|
237
|
+
def call(self, name: str, fn: Callable, *positional, **args):
|
|
238
|
+
"""Call fn(*positional, **args), record it as a tool call (result or error, and timing), and
|
|
239
|
+
return its result. Exceptions are recorded, then re-raised."""
|
|
240
|
+
started = datetime.now(timezone.utc)
|
|
241
|
+
try:
|
|
242
|
+
out = fn(*positional, **args)
|
|
243
|
+
except Exception as e:
|
|
244
|
+
self.tool(name, args, error=f"{type(e).__name__}: {e}"[:2000], started=started,
|
|
245
|
+
ended=datetime.now(timezone.utc))
|
|
246
|
+
raise
|
|
247
|
+
self.tool(name, args, out, started=started, ended=datetime.now(timezone.utc))
|
|
248
|
+
return out
|
|
249
|
+
|
|
250
|
+
def state(self, obj: str, op: str = "update", value: Any = None) -> None:
|
|
251
|
+
"""A change to the world, e.g. state("order:17", "update", {"qty": 3}). op: create, update, delete."""
|
|
252
|
+
self._step("state", name=obj, op=op, value=self._c.clean(value))
|
|
253
|
+
|
|
254
|
+
def answer(self, text: str) -> None:
|
|
255
|
+
self.answer_text = text
|
|
256
|
+
self._step("answer", text=self._c.clean(text))
|
|
257
|
+
|
|
258
|
+
@contextmanager
|
|
259
|
+
def stage(self, name: str, prompt: Optional[str] = None):
|
|
260
|
+
"""A pipeline stage. Record what it produced in .outputs; llm/tool calls inside are nested under it.
|
|
261
|
+
|
|
262
|
+
with run.stage("extract") as s:
|
|
263
|
+
s.outputs.update(extract(text))
|
|
264
|
+
"""
|
|
265
|
+
handle = _Stage()
|
|
266
|
+
started = datetime.now(timezone.utc)
|
|
267
|
+
with self._lock:
|
|
268
|
+
seq, self._seq = self._seq, self._seq + 1
|
|
269
|
+
self._parents.append(seq)
|
|
270
|
+
error = None
|
|
271
|
+
try:
|
|
272
|
+
yield handle
|
|
273
|
+
except Exception as e:
|
|
274
|
+
error = f"{type(e).__name__}: {e}"[:2000]
|
|
275
|
+
raise
|
|
276
|
+
finally:
|
|
277
|
+
with self._lock:
|
|
278
|
+
self._parents.remove(seq)
|
|
279
|
+
if self._on:
|
|
280
|
+
self._c.emit({"type": "step", "run_id": self.id, "seq": seq, "kind": "stage", "name": name,
|
|
281
|
+
"ts": _ts(started), "ended_at": _now(), "status": "error" if error else "ok",
|
|
282
|
+
"error": error, "outputs": self._c.clean(handle.outputs) or None,
|
|
283
|
+
"did_work": handle.did_work, "prompt": prompt})
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
class _Stage:
|
|
287
|
+
def __init__(self):
|
|
288
|
+
self.outputs: Dict[str, Any] = {}
|
|
289
|
+
self.did_work: Optional[bool] = True
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
@contextmanager
|
|
293
|
+
def run(task: Optional[str] = None, *, run_id: Optional[str] = None, kind: str = "agent", input: Any = None,
|
|
294
|
+
input_ref: Optional[str] = None, version: Optional[Dict[str, str]] = None, segment: Optional[str] = None,
|
|
295
|
+
test: Optional[Dict[str, Any]] = None, parent: Optional[Run] = None, tags: Optional[Dict[str, Any]] = None):
|
|
296
|
+
"""Record one run. kind is "agent" (llm/tool/state/answer steps) or "pipeline" (stages).
|
|
297
|
+
test={"run": "nightly-0924", "case": "case-17", "attempt": 0} marks a test-case run.
|
|
298
|
+
An exception inside the block ends the run as failed (and is re-raised)."""
|
|
299
|
+
c = _c()
|
|
300
|
+
rid = run_id or uuid.uuid4().hex
|
|
301
|
+
recorded = c.sample >= 1 or random.random() < c.sample
|
|
302
|
+
r = Run(c, rid, recorded)
|
|
303
|
+
if recorded:
|
|
304
|
+
c.emit({"type": "run.start", "run_id": rid, "kind": kind, "task": task, "segment": segment,
|
|
305
|
+
"input": c.clean(input), "input_ref": input_ref, "version": version, "test": test,
|
|
306
|
+
"parent_run_id": parent.id if parent else None, "tags": tags})
|
|
307
|
+
status, error = "completed", None
|
|
308
|
+
try:
|
|
309
|
+
yield r
|
|
310
|
+
except BaseException as e:
|
|
311
|
+
status, error = "failed", f"{type(e).__name__}: {e}"[:2000]
|
|
312
|
+
raise
|
|
313
|
+
finally:
|
|
314
|
+
if recorded:
|
|
315
|
+
c.emit({"type": "run.end", "run_id": rid, "status": status, "error": error})
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def feedback(run_id: str, kind: str, note: Optional[str] = None) -> None:
|
|
319
|
+
"""What a user did: thumbs_up, thumbs_down, retry, escalation or complaint."""
|
|
320
|
+
_c().emit({"type": "feedback", "run_id": run_id, "kind": kind, "note": note})
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def check(test_run: str, case: str, status: str, *, attempt: Optional[int] = None, run_id: Optional[str] = None,
|
|
324
|
+
field: Optional[str] = None, expected: Any = None, actual: Any = None, evaluator: Optional[str] = None,
|
|
325
|
+
score: Optional[float] = None, reason: Optional[str] = None,
|
|
326
|
+
version: Optional[Dict[str, str]] = None) -> None:
|
|
327
|
+
"""One result from a test run: status pass, fail, or error (the check couldn't run). Send passes too."""
|
|
328
|
+
s = lambda v: None if v is None else v if isinstance(v, str) else json.dumps(_json_safe(v))
|
|
329
|
+
_c().emit({"type": "check", "test": {k: v for k, v in {"run": test_run, "case": case, "attempt": attempt}.items()
|
|
330
|
+
if v is not None},
|
|
331
|
+
"status": status, "run_id": run_id, "field": field, "expected": s(expected), "actual": s(actual),
|
|
332
|
+
"evaluator": evaluator, "score": score, "reason": reason, "version": version})
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def correction(run_id: str, field: str, expected: Any = None, observed: Any = None, kind: str = "wrong",
|
|
336
|
+
reporter: Optional[str] = None) -> None:
|
|
337
|
+
"""Someone found a wrong value: Assay traces it to the step it started at."""
|
|
338
|
+
s = lambda v: None if v is None else str(v)
|
|
339
|
+
_c().emit({"type": "correction", "run_id": run_id, "field": field, "expected": s(expected),
|
|
340
|
+
"observed": s(observed), "kind": kind, "reporter": reporter})
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def expect(case: str, *, calls: Optional[List[Dict[str, Any]]] = None, answer: Optional[str] = None,
|
|
344
|
+
state: Optional[List[Dict[str, Any]]] = None, allow_extra: Optional[List[str]] = None,
|
|
345
|
+
max_steps: Optional[int] = None, answer_match: str = "contains") -> None:
|
|
346
|
+
"""What a test case should do: the tool calls, the answer, and the end state."""
|
|
347
|
+
_c().emit({"type": "expect", "case": case, "calls": calls or [], "answer": answer, "answer_match": answer_match,
|
|
348
|
+
"state": state or [], "allow_extra": allow_extra or [], "max_steps": max_steps})
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def flush() -> bool:
|
|
352
|
+
"""Send everything now (e.g. before a short-lived process exits). True if nothing is left."""
|
|
353
|
+
return _c().flush() if _client else True
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def shutdown() -> None:
|
|
357
|
+
if _client:
|
|
358
|
+
_client.close()
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
atexit.register(lambda: _client and _client.close())
|
|
File without changes
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "assay-evals"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
authors = [{ name = "tap222" }]
|
|
9
|
+
requires-python = ">=3.9"
|
|
10
|
+
dependencies = []
|
|
11
|
+
keywords = ["llm", "agents", "evaluation", "observability", "tracing", "ai"]
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 3 - Alpha",
|
|
14
|
+
"Intended Audience :: Developers",
|
|
15
|
+
"Operating System :: OS Independent",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
"Topic :: Software Development :: Testing",
|
|
20
|
+
"Typing :: Typed",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/tap222/docai-eval"
|
|
25
|
+
Documentation = "https://github.com/tap222/docai-eval/tree/main/sdk/python#readme"
|
|
26
|
+
"Event schema" = "https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md"
|
|
27
|
+
Issues = "https://github.com/tap222/docai-eval/issues"
|
|
28
|
+
|
|
29
|
+
[build-system]
|
|
30
|
+
requires = ["setuptools>=77"]
|
|
31
|
+
build-backend = "setuptools.build_meta"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools]
|
|
34
|
+
packages = ["assay_sdk"]
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.package-data]
|
|
37
|
+
assay_sdk = ["py.typed"]
|