assay-evals 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 tap222
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.4
2
+ Name: assay-evals
3
+ Version: 0.1.0
4
+ Summary: Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results
5
+ Author: tap222
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/tap222/docai-eval
8
+ Project-URL: Documentation, https://github.com/tap222/docai-eval/tree/main/sdk/python#readme
9
+ Project-URL: Event schema, https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md
10
+ Project-URL: Issues, https://github.com/tap222/docai-eval/issues
11
+ Keywords: llm,agents,evaluation,observability,tracing,ai
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Software Development :: Testing
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.9
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Dynamic: license-file
24
+
25
+ # assay-evals
26
+
27
+ Record what your AI system does, and how it went, in
28
+ [Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
29
+ results. Standard library only, Python 3.9+.
30
+
31
+ ```bash
32
+ pip install assay-evals
33
+ ```
34
+
35
+ You need an Assay server to send to. See the
36
+ [setup guide](https://github.com/tap222/docai-eval#setup-guide).
37
+
38
+ ```python
39
+ import assay_sdk as assay
40
+
41
+ assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
42
+
43
+ # An agent
44
+ with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
45
+ run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
46
+ order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
47
+ run.state("refund:O-17", "create", {"amount": order["price"]})
48
+ run.answer(f"Refunded ${order['price']}.")
49
+
50
+ # A pipeline
51
+ with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
52
+ with run.stage("extract", prompt="extract_fields@v13") as s:
53
+ run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
54
+ s.outputs.update(fields)
55
+
56
+ # Outcomes, whenever they're known
57
+ assay.feedback(run.id, "thumbs_down")
58
+ assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
59
+ assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
60
+ assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
61
+ ```
62
+
63
+ | Call | Records |
64
+ |---|---|
65
+ | `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
66
+ | `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
67
+ | `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
68
+ | `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
69
+
70
+ These options go to `init()`:
71
+ - `redact`: a function applied to inputs, arguments, results, text and outputs before they
72
+ leave the process.
73
+ - `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
74
+ are always sent.
75
+ - `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
76
+ your code.
77
+ - `enabled=False`: the SDK does nothing, e.g. in unit tests.
78
+
79
+ Events follow the
80
+ [Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
81
+ `POST /v1/ingest` in the background, so a run that crashes still shows every step up to
82
+ the crash.
@@ -0,0 +1,58 @@
1
+ # assay-evals
2
+
3
+ Record what your AI system does, and how it went, in
4
+ [Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
5
+ results. Standard library only, Python 3.9+.
6
+
7
+ ```bash
8
+ pip install assay-evals
9
+ ```
10
+
11
+ You need an Assay server to send to. See the
12
+ [setup guide](https://github.com/tap222/docai-eval#setup-guide).
13
+
14
+ ```python
15
+ import assay_sdk as assay
16
+
17
+ assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
18
+
19
+ # An agent
20
+ with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
21
+ run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
22
+ order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
23
+ run.state("refund:O-17", "create", {"amount": order["price"]})
24
+ run.answer(f"Refunded ${order['price']}.")
25
+
26
+ # A pipeline
27
+ with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
28
+ with run.stage("extract", prompt="extract_fields@v13") as s:
29
+ run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
30
+ s.outputs.update(fields)
31
+
32
+ # Outcomes, whenever they're known
33
+ assay.feedback(run.id, "thumbs_down")
34
+ assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
35
+ assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
36
+ assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
37
+ ```
38
+
39
+ | Call | Records |
40
+ |---|---|
41
+ | `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
42
+ | `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
43
+ | `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
44
+ | `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
45
+
46
+ These options go to `init()`:
47
+ - `redact`: a function applied to inputs, arguments, results, text and outputs before they
48
+ leave the process.
49
+ - `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
50
+ are always sent.
51
+ - `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
52
+ your code.
53
+ - `enabled=False`: the SDK does nothing, e.g. in unit tests.
54
+
55
+ Events follow the
56
+ [Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
57
+ `POST /v1/ingest` in the background, so a run that crashes still shows every step up to
58
+ the crash.
@@ -0,0 +1,82 @@
1
+ Metadata-Version: 2.4
2
+ Name: assay-evals
3
+ Version: 0.1.0
4
+ Summary: Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results
5
+ Author: tap222
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/tap222/docai-eval
8
+ Project-URL: Documentation, https://github.com/tap222/docai-eval/tree/main/sdk/python#readme
9
+ Project-URL: Event schema, https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md
10
+ Project-URL: Issues, https://github.com/tap222/docai-eval/issues
11
+ Keywords: llm,agents,evaluation,observability,tracing,ai
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Software Development :: Testing
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.9
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Dynamic: license-file
24
+
25
+ # assay-evals
26
+
27
+ Record what your AI system does, and how it went, in
28
+ [Assay](https://github.com/tap222/docai-eval): runs, agent steps, user feedback and test
29
+ results. Standard library only, Python 3.9+.
30
+
31
+ ```bash
32
+ pip install assay-evals
33
+ ```
34
+
35
+ You need an Assay server to send to. See the
36
+ [setup guide](https://github.com/tap222/docai-eval#setup-guide).
37
+
38
+ ```python
39
+ import assay_sdk as assay
40
+
41
+ assay.init("https://assay.example.com", key="ak_...") # or set ASSAY_URL / ASSAY_KEY
42
+
43
+ # An agent
44
+ with assay.run("refund_request", input=message, version={"prompt": "support@v5", "model": "claude-sonnet-5"}) as run:
45
+ run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
46
+ order = run.call("get_order", get_order, order_id="O-17") # runs it; records the result or the error
47
+ run.state("refund:O-17", "create", {"amount": order["price"]})
48
+ run.answer(f"Refunded ${order['price']}.")
49
+
50
+ # A pipeline
51
+ with assay.run("invoice", kind="pipeline", input_ref="s3://inbox/inv-9.pdf") as run:
52
+ with run.stage("extract", prompt="extract_fields@v13") as s:
53
+ run.llm(model="claude-sonnet-5", cost_usd=0.01) # nested under the stage
54
+ s.outputs.update(fields)
55
+
56
+ # Outcomes, whenever they're known
57
+ assay.feedback(run.id, "thumbs_down")
58
+ assay.correction(run.id, "total", expected="1240.00", observed="1204.00")
59
+ assay.check("nightly-0924", "case-17", "fail", run_id=run.id, field="total", expected="1240.00", actual="1204.00")
60
+ assay.expect("case-17", calls=[{"tool": "get_order", "args": {"order_id": "O-17"}}], answer="27.61")
61
+ ```
62
+
63
+ | Call | Records |
64
+ |---|---|
65
+ | `assay.run(task, kind="agent"\|"pipeline", input=, version=, test={"run", "case", "attempt"})` | one run; an exception ends it as failed |
66
+ | `run.llm(...)`, `run.tool(name, args, result)`, `run.call(name, fn, **args)`, `run.state(obj, op, value)`, `run.answer(text)`, `with run.stage(name) as s` | its steps, in order |
67
+ | `assay.feedback`, `assay.check`, `assay.correction`, `assay.expect` | outcomes, sent whenever they're known |
68
+ | `assay.flush()` | send now (short-lived scripts); also happens every second and at exit |
69
+
70
+ These options go to `init()`:
71
+ - `redact`: a function applied to inputs, arguments, results, text and outputs before they
72
+ leave the process.
73
+ - `sample=0.1`: record one run in ten. A run is recorded whole or not at all, and outcomes
74
+ are always sent.
75
+ - `strict=True`: raise send errors while developing. Otherwise the SDK never raises into
76
+ your code.
77
+ - `enabled=False`: the SDK does nothing, e.g. in unit tests.
78
+
79
+ Events follow the
80
+ [Assay event schema v1](https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md). They stream to
81
+ `POST /v1/ingest` in the background, so a run that crashes still shows every step up to
82
+ the crash.
@@ -0,0 +1,9 @@
1
+ LICENSE
2
+ README.md
3
+ pyproject.toml
4
+ assay_evals.egg-info/PKG-INFO
5
+ assay_evals.egg-info/SOURCES.txt
6
+ assay_evals.egg-info/dependency_links.txt
7
+ assay_evals.egg-info/top_level.txt
8
+ assay_sdk/__init__.py
9
+ assay_sdk/py.typed
@@ -0,0 +1 @@
1
+ assay_sdk
@@ -0,0 +1,361 @@
1
+ """Assay SDK: record what your AI system does, and how it went. Standard library only.
2
+
3
+ import assay_sdk as assay
4
+
5
+ assay.init("https://assay.example.com", key="ak_...") # or ASSAY_URL / ASSAY_KEY
6
+
7
+ with assay.run("refund_request", input=message, version={"prompt": "support@v5"}) as run:
8
+ run.llm(model="claude-sonnet-5", tokens_in=620, tokens_out=180, cost_usd=0.0024)
9
+ order = run.call("get_order", get_order, order_id="O-17") # runs it, records result or error
10
+ run.state("refund:O-17", "create", {"amount": 27.61})
11
+ run.answer("Refunded $27.61.")
12
+
13
+ assay.feedback(run.id, "thumbs_up") # later, from your UI
14
+ assay.check("nightly-0924", "case-17", "pass", run_id=run.id, field="answer") # from your tests
15
+
16
+ Events follow the Assay event schema v1 (docs/event-schema.md) and stream to
17
+ POST /v1/ingest in the background: a run that crashes still shows every step up
18
+ to that point. Every event has an id, so retries never duplicate anything. The
19
+ SDK never raises into your code (pass strict=True to init while developing).
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import atexit
24
+ import json
25
+ import logging
26
+ import os
27
+ import random
28
+ import threading
29
+ import time
30
+ import urllib.error
31
+ import urllib.request
32
+ import uuid
33
+ from contextlib import contextmanager
34
+ from datetime import datetime, timezone
35
+ from typing import Any, Callable, Dict, List, Optional
36
+
37
+ __all__ = ["init", "run", "feedback", "check", "correction", "expect", "flush", "shutdown", "Run"]
38
+ __version__ = "0.1.0"
39
+
40
+ log = logging.getLogger("assay_sdk")
41
+ SCHEMA = 1
42
+
43
+
44
+ def _now() -> str:
45
+ return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
46
+
47
+
48
+ def _ts(t: Optional[datetime]) -> Optional[str]:
49
+ if t is None:
50
+ return None
51
+ if t.tzinfo is None:
52
+ t = t.replace(tzinfo=timezone.utc)
53
+ return t.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
54
+
55
+
56
+ def _json_safe(v):
57
+ """Make a value JSON-serializable without failing: unknown objects become their repr."""
58
+ try:
59
+ json.dumps(v)
60
+ return v
61
+ except (TypeError, ValueError):
62
+ if isinstance(v, dict):
63
+ return {str(k): _json_safe(x) for k, x in v.items()}
64
+ if isinstance(v, (list, tuple, set)):
65
+ return [_json_safe(x) for x in v]
66
+ if isinstance(v, datetime):
67
+ return _ts(v)
68
+ return repr(v)
69
+
70
+
71
+ class _Client:
72
+ def __init__(self, url: str, key: Optional[str], tenant: Optional[str], redact: Optional[Callable],
73
+ sample: float, flush_interval: float, batch_size: int, max_queue: int, strict: bool,
74
+ transport: Optional[Callable[[List[dict]], None]], enabled: bool):
75
+ self.url, self.key, self.tenant = url.rstrip("/"), key, tenant
76
+ self.redact, self.sample, self.strict, self.enabled = redact, sample, strict, enabled
77
+ self.batch_size, self.max_queue = batch_size, max_queue
78
+ self._send = transport or self._http
79
+ self._queue: List[dict] = []
80
+ self._lock = threading.Lock()
81
+ self._wake = threading.Event()
82
+ self._stop = False
83
+ self.dropped = 0
84
+ self._thread = threading.Thread(target=self._loop, args=(flush_interval,), name="assay-evals", daemon=True)
85
+ self._thread.start()
86
+
87
+ def emit(self, event: dict) -> None:
88
+ if not self.enabled:
89
+ return
90
+ event = {"v": SCHEMA, "id": uuid.uuid4().hex, "ts": _now(), **event}
91
+ with self._lock:
92
+ if len(self._queue) >= self.max_queue:
93
+ self._queue.pop(0)
94
+ self.dropped += 1
95
+ if self.dropped in (1, 100, 10000):
96
+ log.warning("Assay: queue full, dropped %d events so far (is the server reachable?)", self.dropped)
97
+ self._queue.append({k: v for k, v in event.items() if v is not None})
98
+ full = len(self._queue) >= self.batch_size
99
+ if full:
100
+ self._wake.set()
101
+
102
+ def clean(self, v):
103
+ """Redact (if configured), then make JSON-safe."""
104
+ if v is None:
105
+ return None
106
+ if self.redact:
107
+ try:
108
+ v = self.redact(v)
109
+ except Exception:
110
+ log.exception("Assay: redact() failed; dropping the value rather than sending it unredacted")
111
+ return "<redaction failed>"
112
+ return _json_safe(v)
113
+
114
+ def flush(self) -> bool:
115
+ """Send everything queued. True if the queue is empty afterwards."""
116
+ while True:
117
+ with self._lock:
118
+ batch, self._queue = self._queue[:self.batch_size], self._queue[self.batch_size:]
119
+ if not batch:
120
+ return True
121
+ try:
122
+ self._send(batch)
123
+ except Exception:
124
+ with self._lock: # keep them, in order, for the next try
125
+ self._queue = batch + self._queue
126
+ if self.strict:
127
+ raise
128
+ log.warning("Assay: couldn't send %d events; will retry", len(batch), exc_info=True)
129
+ return False
130
+
131
+ def _loop(self, interval: float) -> None:
132
+ backoff = interval
133
+ while not self._stop:
134
+ self._wake.wait(backoff)
135
+ self._wake.clear()
136
+ ok = self.flush() if not self.strict else self._flush_quiet()
137
+ backoff = interval if ok else min(backoff * 2, 60.0)
138
+
139
+ def _flush_quiet(self) -> bool:
140
+ try:
141
+ return self.flush()
142
+ except Exception:
143
+ return False
144
+
145
+ def close(self) -> None:
146
+ self._stop = True
147
+ self._wake.set()
148
+ self.flush()
149
+
150
+ def _http(self, batch: List[dict]) -> None:
151
+ headers = {"Content-Type": "application/json", "User-Agent": f"assay-evals-python/{__version__}"}
152
+ if self.key:
153
+ headers["Authorization"] = f"Bearer {self.key}"
154
+ if self.tenant:
155
+ headers["X-Tenant"] = self.tenant
156
+ req = urllib.request.Request(self.url + "/v1/ingest", data=json.dumps({"events": batch}).encode(),
157
+ headers=headers, method="POST")
158
+ for attempt in range(3):
159
+ try:
160
+ with urllib.request.urlopen(req, timeout=10) as r:
161
+ r.read()
162
+ return
163
+ except urllib.error.HTTPError as e:
164
+ if e.code == 422: # a malformed event won't improve with retries: log it and drop the batch
165
+ log.error("Assay rejected events: %s", e.read().decode()[:1000])
166
+ return
167
+ if e.code < 500 and e.code != 429 or attempt == 2:
168
+ raise
169
+ except OSError:
170
+ if attempt == 2:
171
+ raise
172
+ time.sleep(0.5 * 2 ** attempt)
173
+
174
+
175
+ _client: Optional[_Client] = None
176
+
177
+
178
+ def init(url: Optional[str] = None, key: Optional[str] = None, *, tenant: Optional[str] = None,
179
+ redact: Optional[Callable[[Any], Any]] = None, sample: float = 1.0, flush_interval: float = 1.0,
180
+ batch_size: int = 500, max_queue: int = 100_000, strict: bool = False,
181
+ transport: Optional[Callable[[List[dict]], None]] = None, enabled: bool = True) -> None:
182
+ """Configure the SDK once, at startup.
183
+
184
+ url / key default to the ASSAY_URL / ASSAY_KEY environment variables
185
+ redact applied to inputs, arguments, results, text and outputs before they leave the process
186
+ sample share of runs to record (0.1 = one in ten); outcomes are always sent
187
+ strict raise send errors instead of logging them (for development)
188
+ enabled False turns the SDK into a no-op (e.g. in unit tests)
189
+ """
190
+ global _client
191
+ if _client is not None:
192
+ _client.close()
193
+ url = url or os.environ.get("ASSAY_URL")
194
+ if not url and transport is None and enabled:
195
+ raise ValueError("Give init() a url, or set ASSAY_URL.")
196
+ _client = _Client(url or "http://localhost", key or os.environ.get("ASSAY_KEY"), tenant, redact, sample,
197
+ flush_interval, batch_size, max_queue, strict, transport, enabled)
198
+
199
+
200
+ def _c() -> _Client:
201
+ if _client is None:
202
+ init()
203
+ return _client
204
+
205
+
206
+ class Run:
207
+ """One run of your system on one input. Use through `assay.run(...)`."""
208
+
209
+ def __init__(self, client: _Client, run_id: str, recorded: bool):
210
+ self.id, self._c, self._on = run_id, client, recorded
211
+ self._seq, self._lock, self._parents = 0, threading.Lock(), []
212
+ self.answer_text: Optional[str] = None
213
+
214
+ def _step(self, kind: str, started: Optional[datetime] = None, **fields) -> int:
215
+ with self._lock:
216
+ seq, self._seq = self._seq, self._seq + 1
217
+ parent = self._parents[-1] if self._parents else None
218
+ if self._on:
219
+ self._c.emit({"type": "step", "run_id": self.id, "seq": seq, "kind": kind, "parent_seq": parent,
220
+ "ts": _ts(started) or _now(), **fields})
221
+ return seq
222
+
223
+ def llm(self, model: Optional[str] = None, tokens_in: Optional[int] = None, tokens_out: Optional[int] = None,
224
+ cost_usd: Optional[float] = None, prompt: Optional[str] = None, text: Optional[str] = None,
225
+ started: Optional[datetime] = None, ended: Optional[datetime] = None, error: Optional[str] = None) -> None:
226
+ """A model call. prompt is "id@version"; text is the output (or a summary of it)."""
227
+ self._step("llm", started, model=model, tokens_in=tokens_in, tokens_out=tokens_out, cost_usd=cost_usd,
228
+ prompt=prompt, text=self._c.clean(text), ended_at=_ts(ended),
229
+ status="error" if error else "ok", error=error)
230
+
231
+ def tool(self, name: str, args: Optional[Dict[str, Any]] = None, result: Any = None, error: Optional[str] = None,
232
+ started: Optional[datetime] = None, ended: Optional[datetime] = None) -> None:
233
+ """A tool call you've already made."""
234
+ self._step("tool", started, name=name, args=self._c.clean(args or {}), result=self._c.clean(result),
235
+ ended_at=_ts(ended), status="error" if error else "ok", error=error)
236
+
237
+ def call(self, name: str, fn: Callable, *positional, **args):
238
+ """Call fn(*positional, **args), record it as a tool call (result or error, and timing), and
239
+ return its result. Exceptions are recorded, then re-raised."""
240
+ started = datetime.now(timezone.utc)
241
+ try:
242
+ out = fn(*positional, **args)
243
+ except Exception as e:
244
+ self.tool(name, args, error=f"{type(e).__name__}: {e}"[:2000], started=started,
245
+ ended=datetime.now(timezone.utc))
246
+ raise
247
+ self.tool(name, args, out, started=started, ended=datetime.now(timezone.utc))
248
+ return out
249
+
250
+ def state(self, obj: str, op: str = "update", value: Any = None) -> None:
251
+ """A change to the world, e.g. state("order:17", "update", {"qty": 3}). op: create, update, delete."""
252
+ self._step("state", name=obj, op=op, value=self._c.clean(value))
253
+
254
+ def answer(self, text: str) -> None:
255
+ self.answer_text = text
256
+ self._step("answer", text=self._c.clean(text))
257
+
258
+ @contextmanager
259
+ def stage(self, name: str, prompt: Optional[str] = None):
260
+ """A pipeline stage. Record what it produced in .outputs; llm/tool calls inside are nested under it.
261
+
262
+ with run.stage("extract") as s:
263
+ s.outputs.update(extract(text))
264
+ """
265
+ handle = _Stage()
266
+ started = datetime.now(timezone.utc)
267
+ with self._lock:
268
+ seq, self._seq = self._seq, self._seq + 1
269
+ self._parents.append(seq)
270
+ error = None
271
+ try:
272
+ yield handle
273
+ except Exception as e:
274
+ error = f"{type(e).__name__}: {e}"[:2000]
275
+ raise
276
+ finally:
277
+ with self._lock:
278
+ self._parents.remove(seq)
279
+ if self._on:
280
+ self._c.emit({"type": "step", "run_id": self.id, "seq": seq, "kind": "stage", "name": name,
281
+ "ts": _ts(started), "ended_at": _now(), "status": "error" if error else "ok",
282
+ "error": error, "outputs": self._c.clean(handle.outputs) or None,
283
+ "did_work": handle.did_work, "prompt": prompt})
284
+
285
+
286
+ class _Stage:
287
+ def __init__(self):
288
+ self.outputs: Dict[str, Any] = {}
289
+ self.did_work: Optional[bool] = True
290
+
291
+
292
+ @contextmanager
293
+ def run(task: Optional[str] = None, *, run_id: Optional[str] = None, kind: str = "agent", input: Any = None,
294
+ input_ref: Optional[str] = None, version: Optional[Dict[str, str]] = None, segment: Optional[str] = None,
295
+ test: Optional[Dict[str, Any]] = None, parent: Optional[Run] = None, tags: Optional[Dict[str, Any]] = None):
296
+ """Record one run. kind is "agent" (llm/tool/state/answer steps) or "pipeline" (stages).
297
+ test={"run": "nightly-0924", "case": "case-17", "attempt": 0} marks a test-case run.
298
+ An exception inside the block ends the run as failed (and is re-raised)."""
299
+ c = _c()
300
+ rid = run_id or uuid.uuid4().hex
301
+ recorded = c.sample >= 1 or random.random() < c.sample
302
+ r = Run(c, rid, recorded)
303
+ if recorded:
304
+ c.emit({"type": "run.start", "run_id": rid, "kind": kind, "task": task, "segment": segment,
305
+ "input": c.clean(input), "input_ref": input_ref, "version": version, "test": test,
306
+ "parent_run_id": parent.id if parent else None, "tags": tags})
307
+ status, error = "completed", None
308
+ try:
309
+ yield r
310
+ except BaseException as e:
311
+ status, error = "failed", f"{type(e).__name__}: {e}"[:2000]
312
+ raise
313
+ finally:
314
+ if recorded:
315
+ c.emit({"type": "run.end", "run_id": rid, "status": status, "error": error})
316
+
317
+
318
+ def feedback(run_id: str, kind: str, note: Optional[str] = None) -> None:
319
+ """What a user did: thumbs_up, thumbs_down, retry, escalation or complaint."""
320
+ _c().emit({"type": "feedback", "run_id": run_id, "kind": kind, "note": note})
321
+
322
+
323
+ def check(test_run: str, case: str, status: str, *, attempt: Optional[int] = None, run_id: Optional[str] = None,
324
+ field: Optional[str] = None, expected: Any = None, actual: Any = None, evaluator: Optional[str] = None,
325
+ score: Optional[float] = None, reason: Optional[str] = None,
326
+ version: Optional[Dict[str, str]] = None) -> None:
327
+ """One result from a test run: status pass, fail, or error (the check couldn't run). Send passes too."""
328
+ s = lambda v: None if v is None else v if isinstance(v, str) else json.dumps(_json_safe(v))
329
+ _c().emit({"type": "check", "test": {k: v for k, v in {"run": test_run, "case": case, "attempt": attempt}.items()
330
+ if v is not None},
331
+ "status": status, "run_id": run_id, "field": field, "expected": s(expected), "actual": s(actual),
332
+ "evaluator": evaluator, "score": score, "reason": reason, "version": version})
333
+
334
+
335
+ def correction(run_id: str, field: str, expected: Any = None, observed: Any = None, kind: str = "wrong",
336
+ reporter: Optional[str] = None) -> None:
337
+ """Someone found a wrong value: Assay traces it to the step it started at."""
338
+ s = lambda v: None if v is None else str(v)
339
+ _c().emit({"type": "correction", "run_id": run_id, "field": field, "expected": s(expected),
340
+ "observed": s(observed), "kind": kind, "reporter": reporter})
341
+
342
+
343
+ def expect(case: str, *, calls: Optional[List[Dict[str, Any]]] = None, answer: Optional[str] = None,
344
+ state: Optional[List[Dict[str, Any]]] = None, allow_extra: Optional[List[str]] = None,
345
+ max_steps: Optional[int] = None, answer_match: str = "contains") -> None:
346
+ """What a test case should do: the tool calls, the answer, and the end state."""
347
+ _c().emit({"type": "expect", "case": case, "calls": calls or [], "answer": answer, "answer_match": answer_match,
348
+ "state": state or [], "allow_extra": allow_extra or [], "max_steps": max_steps})
349
+
350
+
351
+ def flush() -> bool:
352
+ """Send everything now (e.g. before a short-lived process exits). True if nothing is left."""
353
+ return _c().flush() if _client else True
354
+
355
+
356
+ def shutdown() -> None:
357
+ if _client:
358
+ _client.close()
359
+
360
+
361
+ atexit.register(lambda: _client and _client.close())
File without changes
@@ -0,0 +1,37 @@
1
+ [project]
2
+ name = "assay-evals"
3
+ version = "0.1.0"
4
+ description = "Record what your AI system does, and how it went, in Assay: runs, agent steps, feedback and test results"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ authors = [{ name = "tap222" }]
9
+ requires-python = ">=3.9"
10
+ dependencies = []
11
+ keywords = ["llm", "agents", "evaluation", "observability", "tracing", "ai"]
12
+ classifiers = [
13
+ "Development Status :: 3 - Alpha",
14
+ "Intended Audience :: Developers",
15
+ "Operating System :: OS Independent",
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3 :: Only",
18
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
19
+ "Topic :: Software Development :: Testing",
20
+ "Typing :: Typed",
21
+ ]
22
+
23
+ [project.urls]
24
+ Homepage = "https://github.com/tap222/docai-eval"
25
+ Documentation = "https://github.com/tap222/docai-eval/tree/main/sdk/python#readme"
26
+ "Event schema" = "https://github.com/tap222/docai-eval/blob/main/docs/event-schema.md"
27
+ Issues = "https://github.com/tap222/docai-eval/issues"
28
+
29
+ [build-system]
30
+ requires = ["setuptools>=77"]
31
+ build-backend = "setuptools.build_meta"
32
+
33
+ [tool.setuptools]
34
+ packages = ["assay_sdk"]
35
+
36
+ [tool.setuptools.package-data]
37
+ assay_sdk = ["py.typed"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+