nabit 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nabit/__init__.py +57 -0
- nabit/adapters/__init__.py +10 -0
- nabit/adapters/langgraph.py +176 -0
- nabit/checks.py +198 -0
- nabit/core.py +343 -0
- nabit/sinks.py +41 -0
- nabit-0.3.0.dist-info/METADATA +280 -0
- nabit-0.3.0.dist-info/RECORD +10 -0
- nabit-0.3.0.dist-info/WHEEL +4 -0
- nabit-0.3.0.dist-info/licenses/LICENSE +21 -0
nabit/__init__.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""nabit — did it actually work?
|
|
2
|
+
|
|
3
|
+
A dead-simple verification layer for LLM agents. Your agent says it's done;
|
|
4
|
+
nabit checks the real system state and tells you whether that was true — and
|
|
5
|
+
can re-run the action to fix it when it wasn't.
|
|
6
|
+
|
|
7
|
+
Basic usage:
|
|
8
|
+
|
|
9
|
+
from nabit import verify
|
|
10
|
+
from nabit.checks import has_keys
|
|
11
|
+
|
|
12
|
+
@verify(lambda result, ctx: db.exists("customers", result["id"]),
|
|
13
|
+
retries=2) # self-heal: re-run on failure
|
|
14
|
+
def create_customer(name):
|
|
15
|
+
# ... agent / tool does the work, claims success ...
|
|
16
|
+
return {"id": 42, "status": "created"}
|
|
17
|
+
|
|
18
|
+
If the post-condition returns False, nabit records a failed verification,
|
|
19
|
+
re-runs the action up to `retries` times, and (depending on mode) warns or
|
|
20
|
+
raises — so a lying "success" can't slip through silently.
|
|
21
|
+
|
|
22
|
+
What sets it apart: self-heal retries, run-ID correlation, a composable check
|
|
23
|
+
library (nabit.checks), and pluggable sinks — all with zero dependencies.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from .core import (
|
|
27
|
+
verify,
|
|
28
|
+
run,
|
|
29
|
+
summary,
|
|
30
|
+
VerificationError,
|
|
31
|
+
VerificationResult,
|
|
32
|
+
get_results,
|
|
33
|
+
clear_results,
|
|
34
|
+
add_sink,
|
|
35
|
+
clear_sinks,
|
|
36
|
+
Mode,
|
|
37
|
+
)
|
|
38
|
+
from .sinks import jsonl_sink
|
|
39
|
+
|
|
40
|
+
from . import checks
|
|
41
|
+
|
|
42
|
+
__all__ = [
|
|
43
|
+
"verify",
|
|
44
|
+
"run",
|
|
45
|
+
"summary",
|
|
46
|
+
"VerificationError",
|
|
47
|
+
"VerificationResult",
|
|
48
|
+
"get_results",
|
|
49
|
+
"clear_results",
|
|
50
|
+
"add_sink",
|
|
51
|
+
"clear_sinks",
|
|
52
|
+
"jsonl_sink",
|
|
53
|
+
"checks",
|
|
54
|
+
"Mode",
|
|
55
|
+
]
|
|
56
|
+
|
|
57
|
+
__version__ = "0.3.0"
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Optional framework adapters for nabit.
|
|
2
|
+
|
|
3
|
+
Each adapter lives in its own module and lazily imports its framework, so the
|
|
4
|
+
core `nabit` package stays dependency-free. Import the one you need:
|
|
5
|
+
|
|
6
|
+
from nabit.adapters.langgraph import NabitCallback, verify_node
|
|
7
|
+
|
|
8
|
+
Nothing here is imported by `nabit/__init__.py` — installing nabit never pulls
|
|
9
|
+
in LangGraph/LangChain.
|
|
10
|
+
"""
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""LangGraph / LangChain adapter for nabit.
|
|
2
|
+
|
|
3
|
+
Two ways to plug in, depending on how much you want:
|
|
4
|
+
|
|
5
|
+
1. NabitCallback — a passive callback handler. Add it once to your graph and it
|
|
6
|
+
verifies every tool result against the checks you register. Zero restructure.
|
|
7
|
+
Detection only (a callback can't re-run a tool), but you still get recording,
|
|
8
|
+
run-id grouping, sinks, and summaries.
|
|
9
|
+
|
|
10
|
+
from nabit.adapters.langgraph import NabitCallback
|
|
11
|
+
from nabit.checks import truthy, has_keys
|
|
12
|
+
|
|
13
|
+
cb = NabitCallback(checks={
|
|
14
|
+
"create_customer": has_keys("id"),
|
|
15
|
+
"search_hotels": truthy(), # catches the empty-list silent fail
|
|
16
|
+
})
|
|
17
|
+
graph.invoke(state, config={"callbacks": [cb]})
|
|
18
|
+
print(cb.summary())
|
|
19
|
+
|
|
20
|
+
2. verify_node — wrap a single graph node to get the FULL nabit treatment,
|
|
21
|
+
including self-heal retries (a node is just a function, so it can be re-run):
|
|
22
|
+
|
|
23
|
+
from nabit.adapters.langgraph import verify_node
|
|
24
|
+
from nabit.checks import predicate
|
|
25
|
+
|
|
26
|
+
builder.add_node("book", verify_node(
|
|
27
|
+
book_node,
|
|
28
|
+
predicate(lambda update, state: db.has_booking(update["booking_id"])),
|
|
29
|
+
retries=2,
|
|
30
|
+
))
|
|
31
|
+
|
|
32
|
+
This module lazily imports LangChain — if it isn't installed, `verify_node` and
|
|
33
|
+
the plain decorator still work; only the live callback hookup needs it.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import time
|
|
39
|
+
from typing import Any, Callable, Dict, Optional, Union
|
|
40
|
+
|
|
41
|
+
from ..core import (
|
|
42
|
+
Mode,
|
|
43
|
+
VerificationResult,
|
|
44
|
+
_current_run,
|
|
45
|
+
_emit,
|
|
46
|
+
verify as _verify,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
try: # lazy / optional — nabit never hard-depends on LangChain
|
|
50
|
+
from langchain_core.callbacks import BaseCallbackHandler
|
|
51
|
+
_HAS_LC = True
|
|
52
|
+
except Exception: # noqa: BLE001
|
|
53
|
+
BaseCallbackHandler = object # type: ignore[assignment,misc]
|
|
54
|
+
_HAS_LC = False
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
PostCondition = Callable[[Any, dict], bool]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _extract_output(output: Any) -> Any:
|
|
61
|
+
"""LangChain tool output may be a ToolMessage, a str, or a raw value.
|
|
62
|
+
Pull out the payload we should verify."""
|
|
63
|
+
if hasattr(output, "content"):
|
|
64
|
+
return output.content
|
|
65
|
+
return output
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class NabitCallback(BaseCallbackHandler):
|
|
69
|
+
"""A LangChain/LangGraph callback handler that verifies tool results.
|
|
70
|
+
|
|
71
|
+
Register a `checks` mapping of ``tool_name -> post_condition``. After each
|
|
72
|
+
tool runs, the matching post-condition is checked against the tool's output;
|
|
73
|
+
results are recorded through nabit's normal machinery (so get_results(),
|
|
74
|
+
sinks, and run ids all work). Provide `default_check` to verify every tool.
|
|
75
|
+
|
|
76
|
+
Args:
|
|
77
|
+
checks: {tool_name: (output, ctx) -> bool}. ctx carries {"tool", "inputs"}.
|
|
78
|
+
default_check: applied to any tool without a specific entry.
|
|
79
|
+
mode: Mode.WARN (default), LOG, SILENT, or RAISE. Note RAISE inside a
|
|
80
|
+
callback may be swallowed by the framework — prefer WARN/SILENT and
|
|
81
|
+
inspect summary()/get_results().
|
|
82
|
+
run_id: optional id stamped on every result for grouping.
|
|
83
|
+
"""
|
|
84
|
+
|
|
85
|
+
def __init__(
|
|
86
|
+
self,
|
|
87
|
+
checks: Optional[Dict[str, PostCondition]] = None,
|
|
88
|
+
*,
|
|
89
|
+
default_check: Optional[PostCondition] = None,
|
|
90
|
+
mode: Union[Mode, str] = Mode.WARN,
|
|
91
|
+
run_id: Optional[str] = None,
|
|
92
|
+
) -> None:
|
|
93
|
+
self.checks = checks or {}
|
|
94
|
+
self.default_check = default_check
|
|
95
|
+
self.mode = Mode(mode)
|
|
96
|
+
self.run_id = run_id
|
|
97
|
+
self._names: Dict[str, str] = {}
|
|
98
|
+
self._starts: Dict[str, float] = {}
|
|
99
|
+
|
|
100
|
+
# -- LangChain callback hooks --------------------------------------------
|
|
101
|
+
def on_tool_start(self, serialized, input_str, *, run_id=None, **kwargs): # noqa: ANN001
|
|
102
|
+
name = None
|
|
103
|
+
if isinstance(serialized, dict):
|
|
104
|
+
name = serialized.get("name")
|
|
105
|
+
name = name or kwargs.get("name") or "tool"
|
|
106
|
+
key = str(run_id)
|
|
107
|
+
self._names[key] = name
|
|
108
|
+
self._starts[key] = time.perf_counter()
|
|
109
|
+
# stash inputs for the post-condition context
|
|
110
|
+
self._names[key + ":inputs"] = kwargs.get("inputs") or input_str # type: ignore[assignment]
|
|
111
|
+
|
|
112
|
+
def on_tool_end(self, output, *, run_id=None, **kwargs): # noqa: ANN001
|
|
113
|
+
key = str(run_id)
|
|
114
|
+
name = self._names.pop(key, "tool")
|
|
115
|
+
inputs = self._names.pop(key + ":inputs", None)
|
|
116
|
+
start = self._starts.pop(key, None)
|
|
117
|
+
check = self.checks.get(name, self.default_check)
|
|
118
|
+
if check is None:
|
|
119
|
+
return # nothing registered for this tool — ignore it
|
|
120
|
+
self._run_check(name, check, _extract_output(output), inputs, start)
|
|
121
|
+
|
|
122
|
+
# -- shared logic (framework-independent, so it's unit-testable) ----------
|
|
123
|
+
def _run_check(self, name: str, check: PostCondition, output: Any,
|
|
124
|
+
inputs: Any, start: Optional[float]) -> None:
|
|
125
|
+
ctx = {"tool": name, "inputs": inputs}
|
|
126
|
+
try:
|
|
127
|
+
passed = bool(check(output, ctx))
|
|
128
|
+
err = None
|
|
129
|
+
except Exception as exc: # noqa: BLE001
|
|
130
|
+
passed = False
|
|
131
|
+
err = f"postcondition raised: {exc}"
|
|
132
|
+
dur = (time.perf_counter() - start) * 1000 if start else 0.0
|
|
133
|
+
_emit(
|
|
134
|
+
VerificationResult(
|
|
135
|
+
name=name, passed=passed, duration_ms=dur, result=output,
|
|
136
|
+
error=err, attempts=1,
|
|
137
|
+
run_id=self.run_id or _current_run.get(),
|
|
138
|
+
),
|
|
139
|
+
self.mode,
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
def summary(self) -> dict:
|
|
143
|
+
"""Pass/fail summary scoped to this handler's run_id (if set)."""
|
|
144
|
+
from ..core import summary as _summary
|
|
145
|
+
|
|
146
|
+
return _summary(run_id=self.run_id) if self.run_id else _summary()
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def verify_node(
|
|
150
|
+
node: Callable,
|
|
151
|
+
postcondition: PostCondition,
|
|
152
|
+
*,
|
|
153
|
+
mode: Union[Mode, str] = Mode.WARN,
|
|
154
|
+
name: Optional[str] = None,
|
|
155
|
+
retries: int = 0,
|
|
156
|
+
backoff: float = 0.0,
|
|
157
|
+
on_retry: Optional[Callable] = None,
|
|
158
|
+
):
|
|
159
|
+
"""Wrap a LangGraph node function with nabit verification + self-heal.
|
|
160
|
+
|
|
161
|
+
A node is ``(state) -> state_update``. The post-condition receives
|
|
162
|
+
``(state_update, ctx)`` where ctx holds the bound node args (so ctx["state"]
|
|
163
|
+
is the incoming state). Because a node is a plain function, this path gets
|
|
164
|
+
the full retry/self-heal behavior that the passive callback can't offer.
|
|
165
|
+
"""
|
|
166
|
+
return _verify(
|
|
167
|
+
postcondition,
|
|
168
|
+
mode=mode,
|
|
169
|
+
name=name or getattr(node, "__name__", "node"),
|
|
170
|
+
retries=retries,
|
|
171
|
+
backoff=backoff,
|
|
172
|
+
on_retry=on_retry,
|
|
173
|
+
)(node)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
__all__ = ["NabitCallback", "verify_node", "_HAS_LC"]
|
nabit/checks.py
ADDED
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
"""Composable post-condition builders.
|
|
2
|
+
|
|
3
|
+
Most verifications are the same handful of shapes: "the result has these keys",
|
|
4
|
+
"this file now exists", "this URL returns 2xx", "this row is in the DB". Instead
|
|
5
|
+
of hand-writing a lambda every time (what every other verifier makes you do),
|
|
6
|
+
compose these.
|
|
7
|
+
|
|
8
|
+
Every builder returns a post-condition `(result, context) -> bool` that plugs
|
|
9
|
+
straight into @verify:
|
|
10
|
+
|
|
11
|
+
from nabit import verify
|
|
12
|
+
from nabit.checks import has_keys, all_of, file_exists
|
|
13
|
+
|
|
14
|
+
@verify(all_of(has_keys("id", "status"), lambda r, c: r["status"] == "created"))
|
|
15
|
+
def create(...): ...
|
|
16
|
+
|
|
17
|
+
Zero dependencies. `http_ok` uses only the stdlib (urllib).
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import os
|
|
23
|
+
from typing import Any, Callable, Mapping
|
|
24
|
+
|
|
25
|
+
PostCondition = Callable[[Any, dict], bool]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# --- combinators ------------------------------------------------------------
|
|
29
|
+
def all_of(*checks: PostCondition) -> PostCondition:
|
|
30
|
+
"""Pass only if every check passes (logical AND)."""
|
|
31
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
32
|
+
return all(bool(c(result, ctx)) for c in checks)
|
|
33
|
+
return _pc
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def any_of(*checks: PostCondition) -> PostCondition:
|
|
37
|
+
"""Pass if any check passes (logical OR)."""
|
|
38
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
39
|
+
return any(bool(c(result, ctx)) for c in checks)
|
|
40
|
+
return _pc
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def not_(check: PostCondition) -> PostCondition:
|
|
44
|
+
"""Invert a check."""
|
|
45
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
46
|
+
return not bool(check(result, ctx))
|
|
47
|
+
return _pc
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
# --- result-shape checks ----------------------------------------------------
|
|
51
|
+
def has_keys(*keys: str) -> PostCondition:
|
|
52
|
+
"""Result is a mapping containing all of `keys`."""
|
|
53
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
54
|
+
if not isinstance(result, Mapping):
|
|
55
|
+
return False
|
|
56
|
+
return all(k in result for k in keys)
|
|
57
|
+
return _pc
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def equals(expected: Any) -> PostCondition:
|
|
61
|
+
"""Result equals `expected`."""
|
|
62
|
+
return lambda result, ctx: result == expected
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def field_equals(key: str, expected: Any) -> PostCondition:
|
|
66
|
+
"""Result is a mapping where result[key] == expected."""
|
|
67
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
68
|
+
return isinstance(result, Mapping) and result.get(key) == expected
|
|
69
|
+
return _pc
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def truthy(key: str | None = None) -> PostCondition:
|
|
73
|
+
"""Result (or result[key]) is truthy. Catches empty lists/strings/None —
|
|
74
|
+
the classic 'returned [] instead of failing' silent failure."""
|
|
75
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
76
|
+
if key is None:
|
|
77
|
+
return bool(result)
|
|
78
|
+
return isinstance(result, Mapping) and bool(result.get(key))
|
|
79
|
+
return _pc
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def non_empty() -> PostCondition:
|
|
83
|
+
"""Result is non-empty (len > 0). Catches truncated/empty responses."""
|
|
84
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
85
|
+
try:
|
|
86
|
+
return len(result) > 0
|
|
87
|
+
except TypeError:
|
|
88
|
+
return result is not None
|
|
89
|
+
return _pc
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _pluck(result: Any, key: Optional[str]) -> Any:
|
|
93
|
+
"""Get result[key] if key given and result is a mapping/attr, else result."""
|
|
94
|
+
if key is None:
|
|
95
|
+
return result
|
|
96
|
+
if isinstance(result, Mapping):
|
|
97
|
+
return result.get(key)
|
|
98
|
+
return getattr(result, key, None)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def min_length(n: int, key: Optional[str] = None) -> PostCondition:
|
|
102
|
+
"""Result (or result[key]) has len >= n. Catches the 'evidence is just a
|
|
103
|
+
URL' / truncated-output silent failure (e.g. an agent's finding must carry
|
|
104
|
+
>100 chars of real proof, not a one-line claim)."""
|
|
105
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
106
|
+
val = _pluck(result, key)
|
|
107
|
+
try:
|
|
108
|
+
return len(val) >= n
|
|
109
|
+
except TypeError:
|
|
110
|
+
return False
|
|
111
|
+
return _pc
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def contains(needle: Any, key: Optional[str] = None) -> PostCondition:
|
|
115
|
+
"""`needle` is contained in result (or result[key]). Works for substrings,
|
|
116
|
+
list membership, dict keys — anything supporting `in`. Catches 'the proof
|
|
117
|
+
doesn't actually mention the target host' style lies."""
|
|
118
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
119
|
+
val = _pluck(result, key)
|
|
120
|
+
try:
|
|
121
|
+
return needle in val
|
|
122
|
+
except TypeError:
|
|
123
|
+
return False
|
|
124
|
+
return _pc
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def matches(pattern: str, key: Optional[str] = None) -> PostCondition:
|
|
128
|
+
"""A regex search succeeds against result (or result[key]), coerced to str.
|
|
129
|
+
Use for 'evidence must contain a hostname / HTTP status / timestamp'."""
|
|
130
|
+
import re
|
|
131
|
+
|
|
132
|
+
rx = re.compile(pattern)
|
|
133
|
+
|
|
134
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
135
|
+
val = _pluck(result, key)
|
|
136
|
+
return val is not None and bool(rx.search(str(val)))
|
|
137
|
+
return _pc
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def in_range(lo: float, hi: float, key: Optional[str] = None) -> PostCondition:
|
|
141
|
+
"""Numeric result (or result[key]) is within [lo, hi]. Catches values that
|
|
142
|
+
are out of spec (e.g. an LLM judge returning 0s but marking 'passed')."""
|
|
143
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
144
|
+
val = _pluck(result, key)
|
|
145
|
+
try:
|
|
146
|
+
return lo <= val <= hi
|
|
147
|
+
except TypeError:
|
|
148
|
+
return False
|
|
149
|
+
return _pc
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
# --- real-world side-effect checks (the whole point of nabit) ----------------
|
|
153
|
+
def file_exists(path_or_fn: str | Callable[[Any, dict], str]) -> PostCondition:
|
|
154
|
+
"""A file exists on disk. `path_or_fn` is a literal path or a callable
|
|
155
|
+
(result, ctx) -> path, so you can derive the path from the agent's output."""
|
|
156
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
157
|
+
path = path_or_fn(result, ctx) if callable(path_or_fn) else path_or_fn
|
|
158
|
+
return bool(path) and os.path.exists(path)
|
|
159
|
+
return _pc
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def file_fresh(path_or_fn: str | Callable[[Any, dict], str], max_age_s: float) -> PostCondition:
|
|
163
|
+
"""A file exists AND was modified within the last `max_age_s` seconds.
|
|
164
|
+
Catches the 'cron job exited 0 but wrote nothing new' failure."""
|
|
165
|
+
import time
|
|
166
|
+
|
|
167
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
168
|
+
path = path_or_fn(result, ctx) if callable(path_or_fn) else path_or_fn
|
|
169
|
+
if not path or not os.path.exists(path):
|
|
170
|
+
return False
|
|
171
|
+
return (time.time() - os.path.getmtime(path)) <= max_age_s
|
|
172
|
+
return _pc
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def predicate(fn: Callable[[Any, dict], bool]) -> PostCondition:
|
|
176
|
+
"""Wrap an arbitrary check against real state — a DB lookup, API call, etc.
|
|
177
|
+
Just sugar for readability / composition:
|
|
178
|
+
|
|
179
|
+
predicate(lambda r, c: db.exists("customers", r["id"]))
|
|
180
|
+
"""
|
|
181
|
+
return lambda result, ctx: bool(fn(result, ctx))
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def http_ok(url_or_fn: str | Callable[[Any, dict], str], timeout: float = 5.0) -> PostCondition:
|
|
185
|
+
"""An HTTP GET to the URL returns a 2xx status. Stdlib only (urllib).
|
|
186
|
+
`url_or_fn` is a literal URL or (result, ctx) -> url."""
|
|
187
|
+
from urllib.request import urlopen
|
|
188
|
+
|
|
189
|
+
def _pc(result: Any, ctx: dict) -> bool:
|
|
190
|
+
url = url_or_fn(result, ctx) if callable(url_or_fn) else url_or_fn
|
|
191
|
+
if not url:
|
|
192
|
+
return False
|
|
193
|
+
try:
|
|
194
|
+
with urlopen(url, timeout=timeout) as resp: # noqa: S310 — caller-supplied URL
|
|
195
|
+
return 200 <= resp.status < 300
|
|
196
|
+
except Exception: # noqa: BLE001
|
|
197
|
+
return False
|
|
198
|
+
return _pc
|
nabit/core.py
ADDED
|
@@ -0,0 +1,343 @@
|
|
|
1
|
+
"""Core verification primitives for nabit.
|
|
2
|
+
|
|
3
|
+
The whole idea: an LLM agent (or any tool it calls) returns a result that
|
|
4
|
+
*claims* something happened. We don't trust the claim. We run a read-only
|
|
5
|
+
post-condition against the actual system state and record whether reality
|
|
6
|
+
matched the claim.
|
|
7
|
+
|
|
8
|
+
What sets nabit apart from other verifiers:
|
|
9
|
+
* self-heal — on a failed check it can re-run the action (optionally with
|
|
10
|
+
feedback), so it corrects silent failures instead of just reporting them.
|
|
11
|
+
* run-ID correlation — scope a batch of actions under one run so their
|
|
12
|
+
verifications group together (the "nothing shares a run id" problem).
|
|
13
|
+
* pluggable sinks — tee every result to JSONL / OpenTelemetry / your logger
|
|
14
|
+
without nabit taking on a single runtime dependency.
|
|
15
|
+
|
|
16
|
+
Zero dependencies. Works with sync or async functions. Framework-agnostic —
|
|
17
|
+
LangGraph, LangChain, CrewAI, or a plain function all look the same here.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import contextlib
|
|
23
|
+
import contextvars
|
|
24
|
+
import functools
|
|
25
|
+
import inspect
|
|
26
|
+
import logging
|
|
27
|
+
import time
|
|
28
|
+
import uuid
|
|
29
|
+
from dataclasses import asdict, dataclass, field
|
|
30
|
+
from enum import Enum
|
|
31
|
+
from typing import Any, Awaitable, Callable, Optional, Union
|
|
32
|
+
|
|
33
|
+
logger = logging.getLogger("nabit")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class Mode(str, Enum):
|
|
37
|
+
"""What to do when a post-condition fails (after retries are exhausted).
|
|
38
|
+
|
|
39
|
+
RAISE — raise VerificationError (fail loud, good for tests / CI).
|
|
40
|
+
WARN — log at WARNING level and return the result anyway (good for prod
|
|
41
|
+
rollout: you get the signal without changing behavior).
|
|
42
|
+
LOG — log at INFO level only.
|
|
43
|
+
SILENT — record the result but emit nothing (inspect via get_results()).
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
RAISE = "raise"
|
|
47
|
+
WARN = "warn"
|
|
48
|
+
LOG = "log"
|
|
49
|
+
SILENT = "silent"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class VerificationError(AssertionError):
|
|
53
|
+
"""Raised (in Mode.RAISE) when an agent's claimed outcome does not match
|
|
54
|
+
actual system state, after any retries are exhausted."""
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass
|
|
58
|
+
class VerificationResult:
|
|
59
|
+
"""One verification event, kept in an in-memory log for reporting."""
|
|
60
|
+
|
|
61
|
+
name: str
|
|
62
|
+
passed: bool
|
|
63
|
+
duration_ms: float
|
|
64
|
+
result: Any = None
|
|
65
|
+
error: Optional[str] = None
|
|
66
|
+
attempts: int = 1
|
|
67
|
+
run_id: Optional[str] = None
|
|
68
|
+
timestamp: float = field(default_factory=time.time)
|
|
69
|
+
|
|
70
|
+
def to_dict(self) -> dict:
|
|
71
|
+
d = asdict(self)
|
|
72
|
+
# `result` may not be JSON-serializable; stringify defensively.
|
|
73
|
+
try:
|
|
74
|
+
import json
|
|
75
|
+
|
|
76
|
+
json.dumps(d["result"])
|
|
77
|
+
except (TypeError, ValueError):
|
|
78
|
+
d["result"] = repr(d["result"])
|
|
79
|
+
return d
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# --- run-ID correlation -----------------------------------------------------
|
|
83
|
+
# A context var so concurrent/async runs don't clobber each other's run id.
|
|
84
|
+
_current_run: contextvars.ContextVar[Optional[str]] = contextvars.ContextVar(
|
|
85
|
+
"nabit_run_id", default=None
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@contextlib.contextmanager
|
|
90
|
+
def run(run_id: Optional[str] = None):
|
|
91
|
+
"""Scope a batch of verified actions under a single run id, so their
|
|
92
|
+
results group together.
|
|
93
|
+
|
|
94
|
+
with nabit.run() as rid:
|
|
95
|
+
create_customer(...)
|
|
96
|
+
charge_card(...)
|
|
97
|
+
print(nabit.summary(run_id=rid))
|
|
98
|
+
|
|
99
|
+
If `run_id` is omitted a short unique id is generated and yielded.
|
|
100
|
+
"""
|
|
101
|
+
rid = run_id or uuid.uuid4().hex[:12]
|
|
102
|
+
token = _current_run.set(rid)
|
|
103
|
+
try:
|
|
104
|
+
yield rid
|
|
105
|
+
finally:
|
|
106
|
+
_current_run.reset(token)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# --- results log + pluggable sinks ------------------------------------------
|
|
110
|
+
_RESULTS: list[VerificationResult] = []
|
|
111
|
+
_MAX_RESULTS = 1000
|
|
112
|
+
_SINKS: list[Callable[[VerificationResult], None]] = []
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def add_sink(fn: Callable[[VerificationResult], None]) -> None:
|
|
116
|
+
"""Register a callback invoked with every VerificationResult as it happens.
|
|
117
|
+
Use it to tee results to JSONL, OpenTelemetry, Datadog, etc. Zero deps:
|
|
118
|
+
nabit never imports your sink's backend."""
|
|
119
|
+
_SINKS.append(fn)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def clear_sinks() -> None:
|
|
123
|
+
_SINKS.clear()
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def get_results(run_id: Optional[str] = None) -> list[VerificationResult]:
|
|
127
|
+
"""Return recorded verification results (most recent last). Optionally
|
|
128
|
+
filter to a single run id."""
|
|
129
|
+
if run_id is None:
|
|
130
|
+
return list(_RESULTS)
|
|
131
|
+
return [r for r in _RESULTS if r.run_id == run_id]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def clear_results() -> None:
|
|
135
|
+
"""Clear the in-memory results log (useful between tests)."""
|
|
136
|
+
_RESULTS.clear()
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def summary(run_id: Optional[str] = None) -> dict:
|
|
140
|
+
"""Aggregate pass/fail stats, optionally scoped to one run id."""
|
|
141
|
+
rs = get_results(run_id)
|
|
142
|
+
total = len(rs)
|
|
143
|
+
passed = sum(1 for r in rs if r.passed)
|
|
144
|
+
failed = total - passed
|
|
145
|
+
return {
|
|
146
|
+
"total": total,
|
|
147
|
+
"passed": passed,
|
|
148
|
+
"failed": failed,
|
|
149
|
+
"pass_rate": (passed / total) if total else 1.0,
|
|
150
|
+
"failures": [r.name for r in rs if not r.passed],
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _record(result: VerificationResult) -> None:
|
|
155
|
+
_RESULTS.append(result)
|
|
156
|
+
if len(_RESULTS) > _MAX_RESULTS:
|
|
157
|
+
del _RESULTS[0]
|
|
158
|
+
for sink in _SINKS:
|
|
159
|
+
try:
|
|
160
|
+
sink(result)
|
|
161
|
+
except Exception: # noqa: BLE001 — a broken sink must never break the app
|
|
162
|
+
logger.exception("nabit: sink raised; continuing")
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# A post-condition receives (result, context) and returns a truthy value for
|
|
166
|
+
# "reality matches the claim". `context` is the bound call arguments, so you
|
|
167
|
+
# can check the inputs against the outputs. May be sync or async.
|
|
168
|
+
PostCondition = Callable[[Any, dict], Union[bool, Awaitable[bool]]]
|
|
169
|
+
# A retry/feedback hook: (attempt, last_result, context) -> None. Use it to
|
|
170
|
+
# nudge the agent before the next attempt (e.g. append a corrective message).
|
|
171
|
+
RetryHook = Callable[[int, Any, dict], Any]
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _build_context(func: Callable, args: tuple, kwargs: dict) -> dict:
|
|
175
|
+
"""Bind call args to parameter names so post-conditions can read inputs."""
|
|
176
|
+
try:
|
|
177
|
+
bound = inspect.signature(func).bind_partial(*args, **kwargs)
|
|
178
|
+
bound.apply_defaults()
|
|
179
|
+
return dict(bound.arguments)
|
|
180
|
+
except TypeError:
|
|
181
|
+
return {"args": args, "kwargs": kwargs}
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _emit(result: VerificationResult, mode: Mode) -> None:
|
|
185
|
+
"""Record a final result and raise/log according to mode."""
|
|
186
|
+
_record(result)
|
|
187
|
+
if result.passed:
|
|
188
|
+
logger.debug("nabit: %s verified OK (%.1fms, %d attempt(s))",
|
|
189
|
+
result.name, result.duration_ms, result.attempts)
|
|
190
|
+
return
|
|
191
|
+
msg = (f"nabit: {result.name} FAILED verification after {result.attempts} "
|
|
192
|
+
f"attempt(s) — agent claimed success, reality disagreed")
|
|
193
|
+
if result.error:
|
|
194
|
+
msg += f" ({result.error})"
|
|
195
|
+
if mode is Mode.RAISE:
|
|
196
|
+
raise VerificationError(msg)
|
|
197
|
+
elif mode is Mode.WARN:
|
|
198
|
+
logger.warning(msg)
|
|
199
|
+
elif mode is Mode.LOG:
|
|
200
|
+
logger.info(msg)
|
|
201
|
+
# SILENT: recorded only.
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def verify(
|
|
205
|
+
postcondition: PostCondition,
|
|
206
|
+
*,
|
|
207
|
+
mode: Union[Mode, str] = Mode.WARN,
|
|
208
|
+
name: Optional[str] = None,
|
|
209
|
+
on_error: bool = True,
|
|
210
|
+
retries: int = 0,
|
|
211
|
+
backoff: float = 0.0,
|
|
212
|
+
on_retry: Optional[RetryHook] = None,
|
|
213
|
+
) -> Callable:
|
|
214
|
+
"""Decorator: run the wrapped function, then check its claimed outcome
|
|
215
|
+
against actual system state via `postcondition`. On failure, optionally
|
|
216
|
+
self-heal by re-running the action.
|
|
217
|
+
|
|
218
|
+
Args:
|
|
219
|
+
postcondition: callable (result, context) -> bool. Truthy == reality
|
|
220
|
+
matched the claim. May be sync or async.
|
|
221
|
+
mode: what to do once retries are exhausted. See Mode. Default WARN.
|
|
222
|
+
name: label for the result log. Defaults to the function name.
|
|
223
|
+
on_error: if True (default), also run the post-condition when the
|
|
224
|
+
wrapped function raises — so "it threw but the side effect still
|
|
225
|
+
happened" (or vice versa) is handled. On the final failed attempt
|
|
226
|
+
the original exception is re-raised.
|
|
227
|
+
retries: how many times to re-run the action if verification fails
|
|
228
|
+
(0 = no self-heal, just verify once). This is the closed loop:
|
|
229
|
+
detect AND correct.
|
|
230
|
+
backoff: seconds to sleep between retries, multiplied by attempt number
|
|
231
|
+
(linear backoff). Ignored for async if 0.
|
|
232
|
+
on_retry: optional (attempt, last_result, context) -> None hook invoked
|
|
233
|
+
before each retry — use it to feed the discrepancy back to the agent.
|
|
234
|
+
|
|
235
|
+
Works transparently on both sync and async functions.
|
|
236
|
+
"""
|
|
237
|
+
mode = Mode(mode)
|
|
238
|
+
|
|
239
|
+
def decorator(func: Callable) -> Callable:
|
|
240
|
+
label = name or getattr(func, "__name__", "anonymous")
|
|
241
|
+
pc_is_async = inspect.iscoroutinefunction(postcondition)
|
|
242
|
+
func_is_async = inspect.iscoroutinefunction(func)
|
|
243
|
+
|
|
244
|
+
async def _run_pc_async(result: Any, ctx: dict) -> bool:
|
|
245
|
+
out = postcondition(result, ctx)
|
|
246
|
+
if inspect.isawaitable(out):
|
|
247
|
+
out = await out
|
|
248
|
+
return bool(out)
|
|
249
|
+
|
|
250
|
+
def _run_pc_sync(result: Any, ctx: dict) -> bool:
|
|
251
|
+
if pc_is_async:
|
|
252
|
+
raise RuntimeError(
|
|
253
|
+
f"nabit: async post-condition used on sync function '{label}'. "
|
|
254
|
+
"Make the wrapped function async, or the post-condition sync."
|
|
255
|
+
)
|
|
256
|
+
return bool(postcondition(result, ctx))
|
|
257
|
+
|
|
258
|
+
def _finalize(result, passed, pc_err, last_exc, attempt, start):
|
|
259
|
+
"""Record the final attempt, then raise/return per semantics:
|
|
260
|
+
a real exception always propagates (more informative than a
|
|
261
|
+
VerificationError); otherwise _emit handles raise/warn/log/silent."""
|
|
262
|
+
err = pc_err or (f"function raised: {last_exc}" if last_exc else None)
|
|
263
|
+
vr = VerificationResult(
|
|
264
|
+
name=label, passed=passed,
|
|
265
|
+
duration_ms=(time.perf_counter() - start) * 1000,
|
|
266
|
+
result=result, error=err, attempts=attempt,
|
|
267
|
+
run_id=_current_run.get(),
|
|
268
|
+
)
|
|
269
|
+
if last_exc is not None:
|
|
270
|
+
_record(vr)
|
|
271
|
+
if not passed and mode in (Mode.WARN, Mode.LOG):
|
|
272
|
+
logger.warning("nabit: %s FAILED (function raised after %d "
|
|
273
|
+
"attempt(s)): %s", label, attempt, last_exc)
|
|
274
|
+
raise last_exc
|
|
275
|
+
_emit(vr, mode)
|
|
276
|
+
return result
|
|
277
|
+
|
|
278
|
+
if func_is_async:
|
|
279
|
+
@functools.wraps(func)
|
|
280
|
+
async def async_wrapper(*args, **kwargs):
|
|
281
|
+
import asyncio
|
|
282
|
+
|
|
283
|
+
ctx = _build_context(func, args, kwargs)
|
|
284
|
+
start = time.perf_counter()
|
|
285
|
+
result: Any = None
|
|
286
|
+
for attempt in range(1, retries + 2): # 1 initial + `retries`
|
|
287
|
+
last_exc: Optional[BaseException] = None
|
|
288
|
+
try:
|
|
289
|
+
result = await func(*args, **kwargs)
|
|
290
|
+
except Exception as exc: # noqa: BLE001
|
|
291
|
+
last_exc = exc
|
|
292
|
+
result = None
|
|
293
|
+
if not on_error:
|
|
294
|
+
raise # don't verify, just propagate
|
|
295
|
+
try:
|
|
296
|
+
passed = await _run_pc_async(result, ctx)
|
|
297
|
+
pc_err = None
|
|
298
|
+
except Exception as pc_exc: # noqa: BLE001
|
|
299
|
+
passed = False
|
|
300
|
+
pc_err = f"postcondition raised: {pc_exc}"
|
|
301
|
+
if passed or attempt == retries + 1:
|
|
302
|
+
return _finalize(result, passed, pc_err, last_exc, attempt, start)
|
|
303
|
+
logger.debug("nabit: %s attempt %d failed, retrying", label, attempt)
|
|
304
|
+
if on_retry is not None:
|
|
305
|
+
maybe = on_retry(attempt, result, ctx)
|
|
306
|
+
if inspect.isawaitable(maybe):
|
|
307
|
+
await maybe
|
|
308
|
+
if backoff:
|
|
309
|
+
await asyncio.sleep(backoff * attempt)
|
|
310
|
+
|
|
311
|
+
return async_wrapper
|
|
312
|
+
|
|
313
|
+
@functools.wraps(func)
|
|
314
|
+
def sync_wrapper(*args, **kwargs):
|
|
315
|
+
ctx = _build_context(func, args, kwargs)
|
|
316
|
+
start = time.perf_counter()
|
|
317
|
+
result: Any = None
|
|
318
|
+
for attempt in range(1, retries + 2):
|
|
319
|
+
last_exc: Optional[BaseException] = None
|
|
320
|
+
try:
|
|
321
|
+
result = func(*args, **kwargs)
|
|
322
|
+
except Exception as exc: # noqa: BLE001
|
|
323
|
+
last_exc = exc
|
|
324
|
+
result = None
|
|
325
|
+
if not on_error:
|
|
326
|
+
raise
|
|
327
|
+
try:
|
|
328
|
+
passed = _run_pc_sync(result, ctx)
|
|
329
|
+
pc_err = None
|
|
330
|
+
except Exception as pc_exc: # noqa: BLE001
|
|
331
|
+
passed = False
|
|
332
|
+
pc_err = f"postcondition raised: {pc_exc}"
|
|
333
|
+
if passed or attempt == retries + 1:
|
|
334
|
+
return _finalize(result, passed, pc_err, last_exc, attempt, start)
|
|
335
|
+
logger.debug("nabit: %s attempt %d failed, retrying", label, attempt)
|
|
336
|
+
if on_retry is not None:
|
|
337
|
+
on_retry(attempt, result, ctx)
|
|
338
|
+
if backoff:
|
|
339
|
+
time.sleep(backoff * attempt)
|
|
340
|
+
|
|
341
|
+
return sync_wrapper
|
|
342
|
+
|
|
343
|
+
return decorator
|
nabit/sinks.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Built-in result sinks.
|
|
2
|
+
|
|
3
|
+
A sink is just a callable `(VerificationResult) -> None` registered with
|
|
4
|
+
`nabit.add_sink(...)`. These ship in the box; anything fancier (OpenTelemetry,
|
|
5
|
+
Datadog, Kafka) is a three-line function you write and register — nabit never
|
|
6
|
+
imports those backends itself, so the core stays dependency-free.
|
|
7
|
+
|
|
8
|
+
OpenTelemetry example (you provide the dependency, not nabit):
|
|
9
|
+
|
|
10
|
+
from opentelemetry import trace
|
|
11
|
+
tracer = trace.get_tracer("nabit")
|
|
12
|
+
|
|
13
|
+
def otel_sink(r):
|
|
14
|
+
with tracer.start_as_current_span(f"nabit.verify.{r.name}") as span:
|
|
15
|
+
span.set_attribute("nabit.passed", r.passed)
|
|
16
|
+
span.set_attribute("nabit.attempts", r.attempts)
|
|
17
|
+
if r.run_id:
|
|
18
|
+
span.set_attribute("nabit.run_id", r.run_id)
|
|
19
|
+
|
|
20
|
+
nabit.add_sink(otel_sink)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
import threading
|
|
27
|
+
from typing import Callable
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def jsonl_sink(path: str) -> Callable:
|
|
31
|
+
"""Return a sink that appends each VerificationResult as one JSON line to
|
|
32
|
+
`path`. Thread-safe. Register with nabit.add_sink(jsonl_sink("hits.jsonl"))."""
|
|
33
|
+
lock = threading.Lock()
|
|
34
|
+
|
|
35
|
+
def _sink(result) -> None:
|
|
36
|
+
line = json.dumps(result.to_dict(), default=str)
|
|
37
|
+
with lock:
|
|
38
|
+
with open(path, "a", encoding="utf-8") as fh:
|
|
39
|
+
fh.write(line + "\n")
|
|
40
|
+
|
|
41
|
+
return _sink
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: nabit
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: nab your agent's silent failures — verify LLM-agent outcomes against real system state, with self-heal. Zero dependencies.
|
|
5
|
+
Project-URL: Homepage, https://jakegarnier.com/agent-reliability-kit
|
|
6
|
+
Project-URL: Source, https://github.com/jake-garnier/nabit
|
|
7
|
+
Author-email: Jake Garnier <jakegarnier@gmail.com>
|
|
8
|
+
License: MIT License
|
|
9
|
+
|
|
10
|
+
Copyright (c) 2026 Jake Garnier
|
|
11
|
+
|
|
12
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
13
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
14
|
+
in the Software without restriction, including without limitation the rights
|
|
15
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
16
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
17
|
+
furnished to do so, subject to the following conditions:
|
|
18
|
+
|
|
19
|
+
The above copyright notice and this permission notice shall be included in all
|
|
20
|
+
copies or substantial portions of the Software.
|
|
21
|
+
|
|
22
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
23
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
24
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
25
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
26
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
27
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
28
|
+
SOFTWARE.
|
|
29
|
+
License-File: LICENSE
|
|
30
|
+
Keywords: agents,ai,langchain,langgraph,llm,observability,reliability,verification
|
|
31
|
+
Classifier: Development Status :: 4 - Beta
|
|
32
|
+
Classifier: Intended Audience :: Developers
|
|
33
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
34
|
+
Classifier: Programming Language :: Python :: 3
|
|
35
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
36
|
+
Requires-Python: >=3.9
|
|
37
|
+
Provides-Extra: dev
|
|
38
|
+
Requires-Dist: langchain-core>=0.2; extra == 'dev'
|
|
39
|
+
Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
|
|
40
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
41
|
+
Provides-Extra: langgraph
|
|
42
|
+
Requires-Dist: langchain-core>=0.2; extra == 'langgraph'
|
|
43
|
+
Description-Content-Type: text/markdown
|
|
44
|
+
|
|
45
|
+
# nabit — *nab your agent's silent failures*
|
|
46
|
+
|
|
47
|
+
Your LLM agent said it created the customer. Your database says otherwise. You
|
|
48
|
+
found out three days later from a support ticket.
|
|
49
|
+
|
|
50
|
+
**nabit** is a dead-simple verification layer for LLM agents. Your agent claims
|
|
51
|
+
it's done; `nabit` checks the *real system state*, tells you whether that was
|
|
52
|
+
true, and — if it wasn't — **re-runs the action to fix it.** One decorator.
|
|
53
|
+
Zero dependencies. Sync or async.
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from nabit import verify
|
|
57
|
+
|
|
58
|
+
@verify(lambda result, ctx: db.exists("customers", result["id"]),
|
|
59
|
+
retries=2) # self-heal: re-run on failure
|
|
60
|
+
def create_customer(name):
|
|
61
|
+
# agent / tool does the work and claims success
|
|
62
|
+
return {"id": 42, "status": "created"}
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
If the agent returns `{"status": "created"}` but the row isn't in the database,
|
|
66
|
+
`nabit` catches the lie instead of letting a green dashboard hide it — then
|
|
67
|
+
retries the action up to `retries` times before giving up.
|
|
68
|
+
|
|
69
|
+
## How it's different
|
|
70
|
+
|
|
71
|
+
Most tools in this space either detect problems without fixing them, only check
|
|
72
|
+
the *text the model produced* (not whether the real action happened), or make you
|
|
73
|
+
stand up a backend to do it. nabit:
|
|
74
|
+
|
|
75
|
+
- **checks real side effects**, not output schema (vs Guardrails AI / Instructor)
|
|
76
|
+
- **closes the loop** — self-heal retries, not just detection (vs Drift / trace viewers)
|
|
77
|
+
- **verifies inline at runtime**, not in a postmortem (vs agent-coroner)
|
|
78
|
+
- **has zero dependencies and no backend** — it's a decorator, not a platform (vs COGEXT)
|
|
79
|
+
|
|
80
|
+
## Why this exists
|
|
81
|
+
|
|
82
|
+
Trace viewers (Langfuse, LangSmith, Helicone) answer *"what did the agent do?"*
|
|
83
|
+
really well. They don't answer *"was what it did actually correct?"*
|
|
84
|
+
|
|
85
|
+
The expensive failures are **semantic**, not technical: no exception thrown, the
|
|
86
|
+
tool call succeeded, and the output was still wrong. Those look identical to a
|
|
87
|
+
success at the log level. `nabit` is the layer that checks the claim against
|
|
88
|
+
reality.
|
|
89
|
+
|
|
90
|
+
## Install
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
pip install nabit
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Usage
|
|
97
|
+
|
|
98
|
+
### Post-conditions check reality, not the agent's word
|
|
99
|
+
|
|
100
|
+
A post-condition is `(result, context) -> bool`. `result` is what the function
|
|
101
|
+
returned; `context` is the bound call arguments (so you can compare inputs to
|
|
102
|
+
outputs). Return truthy if reality matches the claim.
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
from nabit import verify
|
|
106
|
+
|
|
107
|
+
@verify(lambda result, ctx: ticket_store.is_closed(result["ticket_id"]))
|
|
108
|
+
def close_ticket(ticket_id):
|
|
109
|
+
agent.act(f"close ticket {ticket_id}")
|
|
110
|
+
return {"ticket_id": ticket_id, "status": "closed"}
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Modes — tune how loud failures are
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from nabit import verify, Mode
|
|
117
|
+
|
|
118
|
+
@verify(check, mode=Mode.WARN) # log a warning, keep running (default — safe for prod)
|
|
119
|
+
@verify(check, mode=Mode.RAISE) # raise VerificationError (great for tests / CI)
|
|
120
|
+
@verify(check, mode=Mode.LOG) # info-level log only
|
|
121
|
+
@verify(check, mode=Mode.SILENT) # record only; inspect later
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
### Self-heal — re-run the action when verification fails
|
|
125
|
+
|
|
126
|
+
```python
|
|
127
|
+
def feedback(attempt, last_result, ctx):
|
|
128
|
+
# optional: nudge the agent before the next attempt
|
|
129
|
+
log.warning("verification failed on attempt %d, retrying", attempt)
|
|
130
|
+
|
|
131
|
+
@verify(check, retries=3, backoff=0.5, on_retry=feedback)
|
|
132
|
+
def book_flight(req):
|
|
133
|
+
...
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
`retries` re-runs the whole action up to N times until the post-condition
|
|
137
|
+
passes; `backoff` adds linear delay between attempts; `on_retry` lets you feed
|
|
138
|
+
the discrepancy back to the agent. Detection *and* correction, in one decorator.
|
|
139
|
+
|
|
140
|
+
### Composable checks (no hand-written lambdas)
|
|
141
|
+
|
|
142
|
+
```python
|
|
143
|
+
from nabit import verify
|
|
144
|
+
from nabit.checks import (all_of, has_keys, field_equals, file_fresh, http_ok,
|
|
145
|
+
truthy, min_length, contains, matches, in_range)
|
|
146
|
+
|
|
147
|
+
@verify(all_of(has_keys("id", "status"), field_equals("status", "created")))
|
|
148
|
+
def create(...): ...
|
|
149
|
+
|
|
150
|
+
@verify(file_fresh(lambda r, c: r["path"], max_age_s=60)) # cron wrote a fresh file?
|
|
151
|
+
def nightly_report(): ...
|
|
152
|
+
|
|
153
|
+
@verify(http_ok(lambda r, c: r["url"])) # deployed URL is live?
|
|
154
|
+
def deploy(): ...
|
|
155
|
+
|
|
156
|
+
@verify(truthy()) # not [] / "" / None
|
|
157
|
+
def search(...): ...
|
|
158
|
+
|
|
159
|
+
# an agent's "reportable" finding must carry real proof, not a one-line claim
|
|
160
|
+
@verify(all_of(min_length(100, key="evidence"), matches(r"https?://", key="evidence")))
|
|
161
|
+
def finish_task(finding): ...
|
|
162
|
+
|
|
163
|
+
@verify(in_range(1, 10, key="score")) # LLM judge score in spec
|
|
164
|
+
def judge(...): ...
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Full check list: `all_of` / `any_of` / `not_`, `has_keys`, `equals`,
|
|
168
|
+
`field_equals`, `truthy`, `non_empty`, `min_length`, `contains`, `matches`,
|
|
169
|
+
`in_range`, `file_exists`, `file_fresh`, `http_ok`, `predicate`.
|
|
170
|
+
|
|
171
|
+
### Group verifications under a run id
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from nabit import run, summary
|
|
175
|
+
|
|
176
|
+
with run("signup-flow") as rid:
|
|
177
|
+
create_customer(...)
|
|
178
|
+
charge_card(...)
|
|
179
|
+
|
|
180
|
+
print(summary(run_id=rid))
|
|
181
|
+
# {'total': 2, 'passed': 1, 'failed': 1, 'pass_rate': 0.5, 'failures': ['charge_card']}
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Solves the "nothing shares a run id" problem — all the checks for one logical
|
|
185
|
+
task carry the same id.
|
|
186
|
+
|
|
187
|
+
### Tee results anywhere (pluggable sinks, still zero-dep)
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
import nabit
|
|
191
|
+
from nabit import jsonl_sink
|
|
192
|
+
|
|
193
|
+
nabit.add_sink(jsonl_sink("verifications.jsonl")) # built-in
|
|
194
|
+
|
|
195
|
+
def otel_sink(r): # or your own, 3 lines
|
|
196
|
+
span.set_attribute("nabit.passed", r.passed)
|
|
197
|
+
nabit.add_sink(otel_sink)
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
### LangGraph / LangChain
|
|
201
|
+
|
|
202
|
+
nabit works with any framework via the decorator, but LangGraph users get a
|
|
203
|
+
first-class adapter (lazily imported — installing nabit never pulls in
|
|
204
|
+
LangChain). Two options:
|
|
205
|
+
|
|
206
|
+
**Passive callback** — add it once, verify every tool result, no restructure:
|
|
207
|
+
|
|
208
|
+
```python
|
|
209
|
+
from nabit.adapters.langgraph import NabitCallback
|
|
210
|
+
from nabit.checks import has_keys, truthy
|
|
211
|
+
|
|
212
|
+
cb = NabitCallback(checks={
|
|
213
|
+
"create_customer": has_keys("id"),
|
|
214
|
+
"search_hotels": truthy(), # catches the empty-list silent failure
|
|
215
|
+
})
|
|
216
|
+
graph.invoke(state, config={"callbacks": [cb]})
|
|
217
|
+
print(cb.summary())
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
**Node wrapper** — wrap a node to get full self-heal (a node is just a function,
|
|
221
|
+
so it can be re-run):
|
|
222
|
+
|
|
223
|
+
```python
|
|
224
|
+
from nabit.adapters.langgraph import verify_node
|
|
225
|
+
from nabit.checks import predicate
|
|
226
|
+
|
|
227
|
+
builder.add_node("book", verify_node(
|
|
228
|
+
book_node,
|
|
229
|
+
predicate(lambda update, ctx: db.has_booking(update["booking_id"])),
|
|
230
|
+
retries=2,
|
|
231
|
+
))
|
|
232
|
+
```
|
|
233
|
+
|
|
234
|
+
### Async works the same way
|
|
235
|
+
|
|
236
|
+
```python
|
|
237
|
+
@verify(pc, retries=2) # async funcs + async post-conditions
|
|
238
|
+
async def create_order(cart):
|
|
239
|
+
...
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
### Inspect what happened
|
|
243
|
+
|
|
244
|
+
```python
|
|
245
|
+
from nabit import get_results
|
|
246
|
+
|
|
247
|
+
for r in get_results():
|
|
248
|
+
print(r.name, "PASS" if r.passed else "FAIL",
|
|
249
|
+
f"{r.duration_ms:.0f}ms", f"attempts={r.attempts}", r.error)
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
### Also catches "it threw but the side effect still happened"
|
|
253
|
+
|
|
254
|
+
By default (`on_error=True`) the post-condition runs even when the wrapped
|
|
255
|
+
function raises — so a tool that errors *after* mutating state (or succeeds in
|
|
256
|
+
reality despite throwing) still gets verified. The original exception is
|
|
257
|
+
re-raised after recording.
|
|
258
|
+
|
|
259
|
+
## Scope
|
|
260
|
+
|
|
261
|
+
`nabit` is the **outcome verifier**: it answers *"did this specific action
|
|
262
|
+
actually happen?"* and corrects it when it didn't. It deliberately does **not**
|
|
263
|
+
try to be an observability platform. For *fleet-wide* behavioral monitoring —
|
|
264
|
+
real-time degradation detection across many runs (step-count blowups, token
|
|
265
|
+
spikes, slow drift over hundreds of runs), trajectory snapshot diffing, and a
|
|
266
|
+
dashboard — see **[The Production Agent Reliability Kit](https://jakegarnier.com/agent-reliability-kit)**,
|
|
267
|
+
built on published research
|
|
268
|
+
([SENTINEL](https://github.com/jake-garnier/sentinel), self-supervised anomaly
|
|
269
|
+
detection for LLM agents).
|
|
270
|
+
|
|
271
|
+
Rule of thumb: use `nabit` to verify *one action's* real effect inline; reach
|
|
272
|
+
for the Kit when you need to watch *patterns across runs* over time.
|
|
273
|
+
|
|
274
|
+
## License
|
|
275
|
+
|
|
276
|
+
MIT — see [LICENSE](LICENSE). Use it anywhere, including commercially.
|
|
277
|
+
|
|
278
|
+
---
|
|
279
|
+
|
|
280
|
+
Built by [Jake Garnier](https://jakegarnier.com).
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
nabit/__init__.py,sha256=D8xszU7YpoGstTrqhW6ZsqXBJSWplEgESqFpA1gDvG0,1458
|
|
2
|
+
nabit/checks.py,sha256=2T2ITrojCd0NMsbonlXt7qdDhooSp6zEN4PbW33EU40,7070
|
|
3
|
+
nabit/core.py,sha256=sC5ClyIUH28L7JVFjjR-S2zyTSt6eQOYy0Ag-N-AjUA,13070
|
|
4
|
+
nabit/sinks.py,sha256=qMTdUtLf7-iuRkwL7AZziUNgF0m5TcBehL9vc1m4HQk,1351
|
|
5
|
+
nabit/adapters/__init__.py,sha256=KvQWOT8T84zKPC6a2E992uJeQ6cYyuNZrd62RRmDFAE,369
|
|
6
|
+
nabit/adapters/langgraph.py,sha256=2yD7DqEp0vG0Ji_3_HWae0rRZf6uRWqLSOdLyHnZlPQ,6483
|
|
7
|
+
nabit-0.3.0.dist-info/METADATA,sha256=zne-jGZ_7CjBECrw6THfE4QbnxXW5QDbD81xPnuGOuQ,10220
|
|
8
|
+
nabit-0.3.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
9
|
+
nabit-0.3.0.dist-info/licenses/LICENSE,sha256=P-wcQkV8adB4cgZ4km_ys9Opnln6bRucGHHIcR_N8cM,1069
|
|
10
|
+
nabit-0.3.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jake Garnier
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|