nabit 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nabit-0.3.0/.gitignore +10 -0
- nabit-0.3.0/LICENSE +21 -0
- nabit-0.3.0/PKG-INFO +280 -0
- nabit-0.3.0/README.md +236 -0
- nabit-0.3.0/examples/basic.py +67 -0
- nabit-0.3.0/examples/dogfood_captions.py +105 -0
- nabit-0.3.0/examples/dogfood_hackbot.py +80 -0
- nabit-0.3.0/nabit/__init__.py +57 -0
- nabit-0.3.0/nabit/adapters/__init__.py +10 -0
- nabit-0.3.0/nabit/adapters/langgraph.py +176 -0
- nabit-0.3.0/nabit/checks.py +198 -0
- nabit-0.3.0/nabit/core.py +343 -0
- nabit-0.3.0/nabit/sinks.py +41 -0
- nabit-0.3.0/pyproject.toml +35 -0
- nabit-0.3.0/tests/test_langgraph_adapter.py +115 -0
- nabit-0.3.0/tests/test_verify.py +280 -0
nabit-0.3.0/.gitignore
ADDED
nabit-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jake Garnier
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
nabit-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: nabit
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: nab your agent's silent failures — verify LLM-agent outcomes against real system state, with self-heal. Zero dependencies.
|
|
5
|
+
Project-URL: Homepage, https://jakegarnier.com/agent-reliability-kit
|
|
6
|
+
Project-URL: Source, https://github.com/jake-garnier/nabit
|
|
7
|
+
Author-email: Jake Garnier <jakegarnier@gmail.com>
|
|
8
|
+
License: MIT License
|
|
9
|
+
|
|
10
|
+
Copyright (c) 2026 Jake Garnier
|
|
11
|
+
|
|
12
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
13
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
14
|
+
in the Software without restriction, including without limitation the rights
|
|
15
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
16
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
17
|
+
furnished to do so, subject to the following conditions:
|
|
18
|
+
|
|
19
|
+
The above copyright notice and this permission notice shall be included in all
|
|
20
|
+
copies or substantial portions of the Software.
|
|
21
|
+
|
|
22
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
23
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
24
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
25
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
26
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
27
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
28
|
+
SOFTWARE.
|
|
29
|
+
License-File: LICENSE
|
|
30
|
+
Keywords: agents,ai,langchain,langgraph,llm,observability,reliability,verification
|
|
31
|
+
Classifier: Development Status :: 4 - Beta
|
|
32
|
+
Classifier: Intended Audience :: Developers
|
|
33
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
+
Classifier: Programming Language :: Python :: 3
|
|
35
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
36
|
+
Requires-Python: >=3.9
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: langchain-core>=0.2; extra == 'dev'
|
|
39
|
+
Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
|
|
40
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
41
|
+
Provides-Extra: langgraph
|
|
42
|
+
Requires-Dist: langchain-core>=0.2; extra == 'langgraph'
|
|
43
|
+
Description-Content-Type: text/markdown
|
|
44
|
+
|
|
45
|
+
# nabit — *nab your agent's silent failures*
|
|
46
|
+
|
|
47
|
+
Your LLM agent said it created the customer. Your database says otherwise. You
|
|
48
|
+
found out three days later from a support ticket.
|
|
49
|
+
|
|
50
|
+
**nabit** is a dead-simple verification layer for LLM agents. Your agent claims
|
|
51
|
+
it's done; `nabit` checks the *real system state*, tells you whether that was
|
|
52
|
+
true, and — if it wasn't — **re-runs the action to fix it.** One decorator.
|
|
53
|
+
Zero dependencies. Sync or async.
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from nabit import verify
|
|
57
|
+
|
|
58
|
+
@verify(lambda result, ctx: db.exists("customers", result["id"]),
|
|
59
|
+
retries=2) # self-heal: re-run on failure
|
|
60
|
+
def create_customer(name):
|
|
61
|
+
# agent / tool does the work and claims success
|
|
62
|
+
return {"id": 42, "status": "created"}
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
If the agent returns `{"status": "created"}` but the row isn't in the database,
|
|
66
|
+
`nabit` catches the lie instead of letting a green dashboard hide it — then
|
|
67
|
+
retries the action up to `retries` times before giving up.
|
|
68
|
+
|
|
69
|
+
## How it's different
|
|
70
|
+
|
|
71
|
+
Most tools in this space either detect problems without fixing them, only check
|
|
72
|
+
the *text the model produced* (not whether the real action happened), or make you
|
|
73
|
+
stand up a backend to do it. nabit:
|
|
74
|
+
|
|
75
|
+
- **checks real side effects**, not output schema (vs Guardrails AI / Instructor)
|
|
76
|
+
- **closes the loop** — self-heal retries, not just detection (vs Drift / trace viewers)
|
|
77
|
+
- **verifies inline at runtime**, not in a postmortem (vs agent-coroner)
|
|
78
|
+
- **has zero dependencies and no backend** — it's a decorator, not a platform (vs COGEXT)
|
|
79
|
+
|
|
80
|
+
## Why this exists
|
|
81
|
+
|
|
82
|
+
Trace viewers (Langfuse, LangSmith, Helicone) answer *"what did the agent do?"*
|
|
83
|
+
really well. They don't answer *"was what it did actually correct?"*
|
|
84
|
+
|
|
85
|
+
The expensive failures are **semantic**, not technical: no exception thrown, the
|
|
86
|
+
tool call succeeded, and the output was still wrong. Those look identical to a
|
|
87
|
+
success at the log level. `nabit` is the layer that checks the claim against
|
|
88
|
+
reality.
|
|
89
|
+
|
|
90
|
+
## Install
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
pip install nabit
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Usage
|
|
97
|
+
|
|
98
|
+
### Post-conditions check reality, not the agent's word
|
|
99
|
+
|
|
100
|
+
A post-condition is `(result, context) -> bool`. `result` is what the function
|
|
101
|
+
returned; `context` is the bound call arguments (so you can compare inputs to
|
|
102
|
+
outputs). Return truthy if reality matches the claim.
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
from nabit import verify
|
|
106
|
+
|
|
107
|
+
@verify(lambda result, ctx: ticket_store.is_closed(result["ticket_id"]))
|
|
108
|
+
def close_ticket(ticket_id):
|
|
109
|
+
agent.act(f"close ticket {ticket_id}")
|
|
110
|
+
return {"ticket_id": ticket_id, "status": "closed"}
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Modes — tune how loud failures are
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from nabit import verify, Mode
|
|
117
|
+
|
|
118
|
+
@verify(check, mode=Mode.WARN) # log a warning, keep running (default — safe for prod)
|
|
119
|
+
@verify(check, mode=Mode.RAISE) # raise VerificationError (great for tests / CI)
|
|
120
|
+
@verify(check, mode=Mode.LOG) # info-level log only
|
|
121
|
+
@verify(check, mode=Mode.SILENT) # record only; inspect later
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Self-heal — re-run the action when verification fails
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
def feedback(attempt, last_result, ctx):
|
|
128
|
+
# optional: nudge the agent before the next attempt
|
|
129
|
+
log.warning("verification failed on attempt %d, retrying", attempt)
|
|
130
|
+
|
|
131
|
+
@verify(check, retries=3, backoff=0.5, on_retry=feedback)
|
|
132
|
+
def book_flight(req):
|
|
133
|
+
...
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`retries` re-runs the whole action up to N times until the post-condition
|
|
137
|
+
passes; `backoff` adds linear delay between attempts; `on_retry` lets you feed
|
|
138
|
+
the discrepancy back to the agent. Detection *and* correction, in one decorator.
|
|
139
|
+
|
|
140
|
+
### Composable checks (no hand-written lambdas)
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
from nabit import verify
|
|
144
|
+
from nabit.checks import (all_of, has_keys, field_equals, file_fresh, http_ok,
|
|
145
|
+
truthy, min_length, contains, matches, in_range)
|
|
146
|
+
|
|
147
|
+
@verify(all_of(has_keys("id", "status"), field_equals("status", "created")))
|
|
148
|
+
def create(...): ...
|
|
149
|
+
|
|
150
|
+
@verify(file_fresh(lambda r, c: r["path"], max_age_s=60)) # cron wrote a fresh file?
|
|
151
|
+
def nightly_report(): ...
|
|
152
|
+
|
|
153
|
+
@verify(http_ok(lambda r, c: r["url"])) # deployed URL is live?
|
|
154
|
+
def deploy(): ...
|
|
155
|
+
|
|
156
|
+
@verify(truthy()) # not [] / "" / None
|
|
157
|
+
def search(...): ...
|
|
158
|
+
|
|
159
|
+
# an agent's "reportable" finding must carry real proof, not a one-line claim
|
|
160
|
+
@verify(all_of(min_length(100, key="evidence"), matches(r"https?://", key="evidence")))
|
|
161
|
+
def finish_task(finding): ...
|
|
162
|
+
|
|
163
|
+
@verify(in_range(1, 10, key="score")) # LLM judge score in spec
|
|
164
|
+
def judge(...): ...
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Full check list: `all_of` / `any_of` / `not_`, `has_keys`, `equals`,
|
|
168
|
+
`field_equals`, `truthy`, `non_empty`, `min_length`, `contains`, `matches`,
|
|
169
|
+
`in_range`, `file_exists`, `file_fresh`, `http_ok`, `predicate`.
|
|
170
|
+
|
|
171
|
+
### Group verifications under a run id
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from nabit import run, summary
|
|
175
|
+
|
|
176
|
+
with run("signup-flow") as rid:
|
|
177
|
+
create_customer(...)
|
|
178
|
+
charge_card(...)
|
|
179
|
+
|
|
180
|
+
print(summary(run_id=rid))
|
|
181
|
+
# {'total': 2, 'passed': 1, 'failed': 1, 'pass_rate': 0.5, 'failures': ['charge_card']}
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Solves the "nothing shares a run id" problem — all the checks for one logical
|
|
185
|
+
task carry the same id.
|
|
186
|
+
|
|
187
|
+
### Tee results anywhere (pluggable sinks, still zero-dep)
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
import nabit
|
|
191
|
+
from nabit import jsonl_sink
|
|
192
|
+
|
|
193
|
+
nabit.add_sink(jsonl_sink("verifications.jsonl")) # built-in
|
|
194
|
+
|
|
195
|
+
def otel_sink(r): # or your own, 3 lines
|
|
196
|
+
span.set_attribute("nabit.passed", r.passed)
|
|
197
|
+
nabit.add_sink(otel_sink)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
### LangGraph / LangChain
|
|
201
|
+
|
|
202
|
+
nabit works with any framework via the decorator, but LangGraph users get a
|
|
203
|
+
first-class adapter (lazily imported — installing nabit never pulls in
|
|
204
|
+
LangChain). Two options:
|
|
205
|
+
|
|
206
|
+
**Passive callback** — add it once, verify every tool result, no restructure:
|
|
207
|
+
|
|
208
|
+
```python
|
|
209
|
+
from nabit.adapters.langgraph import NabitCallback
|
|
210
|
+
from nabit.checks import has_keys, truthy
|
|
211
|
+
|
|
212
|
+
cb = NabitCallback(checks={
|
|
213
|
+
"create_customer": has_keys("id"),
|
|
214
|
+
"search_hotels": truthy(), # catches the empty-list silent failure
|
|
215
|
+
})
|
|
216
|
+
graph.invoke(state, config={"callbacks": [cb]})
|
|
217
|
+
print(cb.summary())
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
**Node wrapper** — wrap a node to get full self-heal (a node is just a function,
|
|
221
|
+
so it can be re-run):
|
|
222
|
+
|
|
223
|
+
```python
|
|
224
|
+
from nabit.adapters.langgraph import verify_node
|
|
225
|
+
from nabit.checks import predicate
|
|
226
|
+
|
|
227
|
+
builder.add_node("book", verify_node(
|
|
228
|
+
book_node,
|
|
229
|
+
predicate(lambda update, ctx: db.has_booking(update["booking_id"])),
|
|
230
|
+
retries=2,
|
|
231
|
+
))
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
### Async works the same way
|
|
235
|
+
|
|
236
|
+
```python
|
|
237
|
+
@verify(pc, retries=2) # async funcs + async post-conditions
|
|
238
|
+
async def create_order(cart):
|
|
239
|
+
...
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
### Inspect what happened
|
|
243
|
+
|
|
244
|
+
```python
|
|
245
|
+
from nabit import get_results
|
|
246
|
+
|
|
247
|
+
for r in get_results():
|
|
248
|
+
print(r.name, "PASS" if r.passed else "FAIL",
|
|
249
|
+
f"{r.duration_ms:.0f}ms", f"attempts={r.attempts}", r.error)
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
### Also catches "it threw but the side effect still happened"
|
|
253
|
+
|
|
254
|
+
By default (`on_error=True`) the post-condition runs even when the wrapped
|
|
255
|
+
function raises — so a tool that errors *after* mutating state (or succeeds in
|
|
256
|
+
reality despite throwing) still gets verified. The original exception is
|
|
257
|
+
re-raised after recording.
|
|
258
|
+
|
|
259
|
+
## Scope
|
|
260
|
+
|
|
261
|
+
`nabit` is the **outcome verifier**: it answers *"did this specific action
|
|
262
|
+
actually happen?"* and corrects it when it didn't. It deliberately does **not**
|
|
263
|
+
try to be an observability platform. For *fleet-wide* behavioral monitoring —
|
|
264
|
+
real-time degradation detection across many runs (step-count blowups, token
|
|
265
|
+
spikes, slow drift over hundreds of runs), trajectory snapshot diffing, and a
|
|
266
|
+
dashboard — see **[The Production Agent Reliability Kit](https://jakegarnier.com/agent-reliability-kit)**,
|
|
267
|
+
built on published research
|
|
268
|
+
([SENTINEL](https://github.com/jake-garnier/sentinel), self-supervised anomaly
|
|
269
|
+
detection for LLM agents).
|
|
270
|
+
|
|
271
|
+
Rule of thumb: use `nabit` to verify *one action's* real effect inline; reach
|
|
272
|
+
for the Kit when you need to watch *patterns across runs* over time.
|
|
273
|
+
|
|
274
|
+
## License
|
|
275
|
+
|
|
276
|
+
MIT — see [LICENSE](LICENSE). Use it anywhere, including commercially.
|
|
277
|
+
|
|
278
|
+
---
|
|
279
|
+
|
|
280
|
+
Built by [Jake Garnier](https://jakegarnier.com).
|
nabit-0.3.0/README.md
ADDED
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
# nabit — *nab your agent's silent failures*
|
|
2
|
+
|
|
3
|
+
Your LLM agent said it created the customer. Your database says otherwise. You
|
|
4
|
+
found out three days later from a support ticket.
|
|
5
|
+
|
|
6
|
+
**nabit** is a dead-simple verification layer for LLM agents. Your agent claims
|
|
7
|
+
it's done; `nabit` checks the *real system state*, tells you whether that was
|
|
8
|
+
true, and — if it wasn't — **re-runs the action to fix it.** One decorator.
|
|
9
|
+
Zero dependencies. Sync or async.
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from nabit import verify
|
|
13
|
+
|
|
14
|
+
@verify(lambda result, ctx: db.exists("customers", result["id"]),
|
|
15
|
+
retries=2) # self-heal: re-run on failure
|
|
16
|
+
def create_customer(name):
|
|
17
|
+
# agent / tool does the work and claims success
|
|
18
|
+
return {"id": 42, "status": "created"}
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
If the agent returns `{"status": "created"}` but the row isn't in the database,
|
|
22
|
+
`nabit` catches the lie instead of letting a green dashboard hide it — then
|
|
23
|
+
retries the action up to `retries` times before giving up.
|
|
24
|
+
|
|
25
|
+
## How it's different
|
|
26
|
+
|
|
27
|
+
Most tools in this space either detect problems without fixing them, only check
|
|
28
|
+
the *text the model produced* (not whether the real action happened), or make you
|
|
29
|
+
stand up a backend to do it. nabit:
|
|
30
|
+
|
|
31
|
+
- **checks real side effects**, not output schema (vs Guardrails AI / Instructor)
|
|
32
|
+
- **closes the loop** — self-heal retries, not just detection (vs Drift / trace viewers)
|
|
33
|
+
- **verifies inline at runtime**, not in a postmortem (vs agent-coroner)
|
|
34
|
+
- **has zero dependencies and no backend** — it's a decorator, not a platform (vs COGEXT)
|
|
35
|
+
|
|
36
|
+
## Why this exists
|
|
37
|
+
|
|
38
|
+
Trace viewers (Langfuse, LangSmith, Helicone) answer *"what did the agent do?"*
|
|
39
|
+
really well. They don't answer *"was what it did actually correct?"*
|
|
40
|
+
|
|
41
|
+
The expensive failures are **semantic**, not technical: no exception thrown, the
|
|
42
|
+
tool call succeeded, and the output was still wrong. Those look identical to a
|
|
43
|
+
success at the log level. `nabit` is the layer that checks the claim against
|
|
44
|
+
reality.
|
|
45
|
+
|
|
46
|
+
## Install
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install nabit
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Usage
|
|
53
|
+
|
|
54
|
+
### Post-conditions check reality, not the agent's word
|
|
55
|
+
|
|
56
|
+
A post-condition is `(result, context) -> bool`. `result` is what the function
|
|
57
|
+
returned; `context` is the bound call arguments (so you can compare inputs to
|
|
58
|
+
outputs). Return truthy if reality matches the claim.
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
from nabit import verify
|
|
62
|
+
|
|
63
|
+
@verify(lambda result, ctx: ticket_store.is_closed(result["ticket_id"]))
|
|
64
|
+
def close_ticket(ticket_id):
|
|
65
|
+
agent.act(f"close ticket {ticket_id}")
|
|
66
|
+
return {"ticket_id": ticket_id, "status": "closed"}
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
### Modes — tune how loud failures are
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from nabit import verify, Mode
|
|
73
|
+
|
|
74
|
+
@verify(check, mode=Mode.WARN) # log a warning, keep running (default — safe for prod)
|
|
75
|
+
@verify(check, mode=Mode.RAISE) # raise VerificationError (great for tests / CI)
|
|
76
|
+
@verify(check, mode=Mode.LOG) # info-level log only
|
|
77
|
+
@verify(check, mode=Mode.SILENT) # record only; inspect later
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
### Self-heal — re-run the action when verification fails
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
def feedback(attempt, last_result, ctx):
|
|
84
|
+
# optional: nudge the agent before the next attempt
|
|
85
|
+
log.warning("verification failed on attempt %d, retrying", attempt)
|
|
86
|
+
|
|
87
|
+
@verify(check, retries=3, backoff=0.5, on_retry=feedback)
|
|
88
|
+
def book_flight(req):
|
|
89
|
+
...
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
`retries` re-runs the whole action up to N times until the post-condition
|
|
93
|
+
passes; `backoff` adds linear delay between attempts; `on_retry` lets you feed
|
|
94
|
+
the discrepancy back to the agent. Detection *and* correction, in one decorator.
|
|
95
|
+
|
|
96
|
+
### Composable checks (no hand-written lambdas)
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
from nabit import verify
|
|
100
|
+
from nabit.checks import (all_of, has_keys, field_equals, file_fresh, http_ok,
|
|
101
|
+
truthy, min_length, contains, matches, in_range)
|
|
102
|
+
|
|
103
|
+
@verify(all_of(has_keys("id", "status"), field_equals("status", "created")))
|
|
104
|
+
def create(...): ...
|
|
105
|
+
|
|
106
|
+
@verify(file_fresh(lambda r, c: r["path"], max_age_s=60)) # cron wrote a fresh file?
|
|
107
|
+
def nightly_report(): ...
|
|
108
|
+
|
|
109
|
+
@verify(http_ok(lambda r, c: r["url"])) # deployed URL is live?
|
|
110
|
+
def deploy(): ...
|
|
111
|
+
|
|
112
|
+
@verify(truthy()) # not [] / "" / None
|
|
113
|
+
def search(...): ...
|
|
114
|
+
|
|
115
|
+
# an agent's "reportable" finding must carry real proof, not a one-line claim
|
|
116
|
+
@verify(all_of(min_length(100, key="evidence"), matches(r"https?://", key="evidence")))
|
|
117
|
+
def finish_task(finding): ...
|
|
118
|
+
|
|
119
|
+
@verify(in_range(1, 10, key="score")) # LLM judge score in spec
|
|
120
|
+
def judge(...): ...
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
Full check list: `all_of` / `any_of` / `not_`, `has_keys`, `equals`,
|
|
124
|
+
`field_equals`, `truthy`, `non_empty`, `min_length`, `contains`, `matches`,
|
|
125
|
+
`in_range`, `file_exists`, `file_fresh`, `http_ok`, `predicate`.
|
|
126
|
+
|
|
127
|
+
### Group verifications under a run id
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
from nabit import run, summary
|
|
131
|
+
|
|
132
|
+
with run("signup-flow") as rid:
|
|
133
|
+
create_customer(...)
|
|
134
|
+
charge_card(...)
|
|
135
|
+
|
|
136
|
+
print(summary(run_id=rid))
|
|
137
|
+
# {'total': 2, 'passed': 1, 'failed': 1, 'pass_rate': 0.5, 'failures': ['charge_card']}
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
Solves the "nothing shares a run id" problem — all the checks for one logical
|
|
141
|
+
task carry the same id.
|
|
142
|
+
|
|
143
|
+
### Tee results anywhere (pluggable sinks, still zero-dep)
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
import nabit
|
|
147
|
+
from nabit import jsonl_sink
|
|
148
|
+
|
|
149
|
+
nabit.add_sink(jsonl_sink("verifications.jsonl")) # built-in
|
|
150
|
+
|
|
151
|
+
def otel_sink(r): # or your own, 3 lines
|
|
152
|
+
span.set_attribute("nabit.passed", r.passed)
|
|
153
|
+
nabit.add_sink(otel_sink)
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### LangGraph / LangChain
|
|
157
|
+
|
|
158
|
+
nabit works with any framework via the decorator, but LangGraph users get a
|
|
159
|
+
first-class adapter (lazily imported — installing nabit never pulls in
|
|
160
|
+
LangChain). Two options:
|
|
161
|
+
|
|
162
|
+
**Passive callback** — add it once, verify every tool result, no restructure:
|
|
163
|
+
|
|
164
|
+
```python
|
|
165
|
+
from nabit.adapters.langgraph import NabitCallback
|
|
166
|
+
from nabit.checks import has_keys, truthy
|
|
167
|
+
|
|
168
|
+
cb = NabitCallback(checks={
|
|
169
|
+
"create_customer": has_keys("id"),
|
|
170
|
+
"search_hotels": truthy(), # catches the empty-list silent failure
|
|
171
|
+
})
|
|
172
|
+
graph.invoke(state, config={"callbacks": [cb]})
|
|
173
|
+
print(cb.summary())
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
**Node wrapper** — wrap a node to get full self-heal (a node is just a function,
|
|
177
|
+
so it can be re-run):
|
|
178
|
+
|
|
179
|
+
```python
|
|
180
|
+
from nabit.adapters.langgraph import verify_node
|
|
181
|
+
from nabit.checks import predicate
|
|
182
|
+
|
|
183
|
+
builder.add_node("book", verify_node(
|
|
184
|
+
book_node,
|
|
185
|
+
predicate(lambda update, ctx: db.has_booking(update["booking_id"])),
|
|
186
|
+
retries=2,
|
|
187
|
+
))
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
### Async works the same way
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
@verify(pc, retries=2) # async funcs + async post-conditions
|
|
194
|
+
async def create_order(cart):
|
|
195
|
+
...
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
### Inspect what happened
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
from nabit import get_results
|
|
202
|
+
|
|
203
|
+
for r in get_results():
|
|
204
|
+
print(r.name, "PASS" if r.passed else "FAIL",
|
|
205
|
+
f"{r.duration_ms:.0f}ms", f"attempts={r.attempts}", r.error)
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
### Also catches "it threw but the side effect still happened"
|
|
209
|
+
|
|
210
|
+
By default (`on_error=True`) the post-condition runs even when the wrapped
|
|
211
|
+
function raises — so a tool that errors *after* mutating state (or succeeds in
|
|
212
|
+
reality despite throwing) still gets verified. The original exception is
|
|
213
|
+
re-raised after recording.
|
|
214
|
+
|
|
215
|
+
## Scope
|
|
216
|
+
|
|
217
|
+
`nabit` is the **outcome verifier**: it answers *"did this specific action
|
|
218
|
+
actually happen?"* and corrects it when it didn't. It deliberately does **not**
|
|
219
|
+
try to be an observability platform. For *fleet-wide* behavioral monitoring —
|
|
220
|
+
real-time degradation detection across many runs (step-count blowups, token
|
|
221
|
+
spikes, slow drift over hundreds of runs), trajectory snapshot diffing, and a
|
|
222
|
+
dashboard — see **[The Production Agent Reliability Kit](https://jakegarnier.com/agent-reliability-kit)**,
|
|
223
|
+
built on published research
|
|
224
|
+
([SENTINEL](https://github.com/jake-garnier/sentinel), self-supervised anomaly
|
|
225
|
+
detection for LLM agents).
|
|
226
|
+
|
|
227
|
+
Rule of thumb: use `nabit` to verify *one action's* real effect inline; reach
|
|
228
|
+
for the Kit when you need to watch *patterns across runs* over time.
|
|
229
|
+
|
|
230
|
+
## License
|
|
231
|
+
|
|
232
|
+
MIT — see [LICENSE](LICENSE). Use it anywhere, including commercially.
|
|
233
|
+
|
|
234
|
+
---
|
|
235
|
+
|
|
236
|
+
Built by [Jake Garnier](https://jakegarnier.com).
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Runnable tour of nabit.
|
|
2
|
+
|
|
3
|
+
python examples/basic.py
|
|
4
|
+
|
|
5
|
+
Shows: catching a silent failure, self-healing a flaky action, grouping
|
|
6
|
+
verifications under a run id, and the composable check library.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import logging
|
|
10
|
+
|
|
11
|
+
import nabit
|
|
12
|
+
from nabit import verify, run, summary, Mode, get_results
|
|
13
|
+
from nabit.checks import all_of, has_keys, field_equals
|
|
14
|
+
|
|
15
|
+
logging.basicConfig(level=logging.WARNING, format="%(message)s")
|
|
16
|
+
|
|
17
|
+
DB = set() # a fake database so the example is self-contained
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# ---- 1. Catch a silent failure ---------------------------------------------
|
|
21
|
+
@verify(lambda result, ctx: result["id"] in DB, mode=Mode.WARN)
|
|
22
|
+
def create_customer_lying(name):
|
|
23
|
+
cid = abs(hash(name)) % 1000
|
|
24
|
+
# oops — forgot to write to DB, but still reports success
|
|
25
|
+
return {"id": cid, "status": "created"}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# ---- 2. Self-heal: a flaky action that succeeds on retry --------------------
|
|
29
|
+
_attempts = {"n": 0}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@verify(lambda result, ctx: result["id"] in DB, mode=Mode.WARN, retries=3)
|
|
33
|
+
def create_customer_flaky(name):
|
|
34
|
+
_attempts["n"] += 1
|
|
35
|
+
cid = abs(hash(name)) % 1000
|
|
36
|
+
if _attempts["n"] >= 2: # fails first time, writes on the 2nd attempt
|
|
37
|
+
DB.add(cid)
|
|
38
|
+
return {"id": cid, "status": "created"}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
# ---- 3. Composable checks instead of hand-written lambdas -------------------
|
|
42
|
+
@verify(all_of(has_keys("id", "status"), field_equals("status", "created")),
|
|
43
|
+
mode=Mode.SILENT)
|
|
44
|
+
def create_customer_shaped(name):
|
|
45
|
+
cid = abs(hash(name)) % 1000
|
|
46
|
+
DB.add(cid)
|
|
47
|
+
return {"id": cid, "status": "created"}
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
if __name__ == "__main__":
|
|
51
|
+
with run("signup-batch") as rid:
|
|
52
|
+
print("1. lying tool -> nabit should flag it")
|
|
53
|
+
create_customer_lying("Ada Lovelace")
|
|
54
|
+
|
|
55
|
+
print("2. flaky tool -> nabit should self-heal and pass")
|
|
56
|
+
create_customer_flaky("Alan Turing")
|
|
57
|
+
|
|
58
|
+
print("3. shaped check -> composable, no lambda")
|
|
59
|
+
create_customer_shaped("Grace Hopper")
|
|
60
|
+
|
|
61
|
+
print("\n--- verification log ---")
|
|
62
|
+
for r in get_results(run_id=rid):
|
|
63
|
+
status = "PASS" if r.passed else "FAIL <-- silent failure caught"
|
|
64
|
+
print(f" {r.name:26} {status:34} attempts={r.attempts}")
|
|
65
|
+
|
|
66
|
+
print("\n--- run summary ---")
|
|
67
|
+
print(" ", summary(run_id=rid))
|