nabit 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nabit/__init__.py ADDED
@@ -0,0 +1,57 @@
1
+ """nabit — did it actually work?
2
+
3
+ A dead-simple verification layer for LLM agents. Your agent says it's done;
4
+ nabit checks the real system state and tells you whether that was true — and
5
+ can re-run the action to fix it when it wasn't.
6
+
7
+ Basic usage:
8
+
9
+ from nabit import verify
10
+ from nabit.checks import has_keys
11
+
12
+ @verify(lambda result, ctx: db.exists("customers", result["id"]),
13
+ retries=2) # self-heal: re-run on failure
14
+ def create_customer(name):
15
+ # ... agent / tool does the work, claims success ...
16
+ return {"id": 42, "status": "created"}
17
+
18
+ If the post-condition returns False, nabit records a failed verification,
19
+ re-runs the action up to `retries` times, and (depending on mode) warns or
20
+ raises — so a lying "success" can't slip through silently.
21
+
22
+ What sets it apart: self-heal retries, run-ID correlation, a composable check
23
+ library (nabit.checks), and pluggable sinks — all with zero dependencies.
24
+ """
25
+
26
+ from .core import (
27
+ verify,
28
+ run,
29
+ summary,
30
+ VerificationError,
31
+ VerificationResult,
32
+ get_results,
33
+ clear_results,
34
+ add_sink,
35
+ clear_sinks,
36
+ Mode,
37
+ )
38
+ from .sinks import jsonl_sink
39
+
40
+ from . import checks
41
+
42
+ __all__ = [
43
+ "verify",
44
+ "run",
45
+ "summary",
46
+ "VerificationError",
47
+ "VerificationResult",
48
+ "get_results",
49
+ "clear_results",
50
+ "add_sink",
51
+ "clear_sinks",
52
+ "jsonl_sink",
53
+ "checks",
54
+ "Mode",
55
+ ]
56
+
57
+ __version__ = "0.3.0"
@@ -0,0 +1,10 @@
1
+ """Optional framework adapters for nabit.
2
+
3
+ Each adapter lives in its own module and lazily imports its framework, so the
4
+ core `nabit` package stays dependency-free. Import the one you need:
5
+
6
+ from nabit.adapters.langgraph import NabitCallback, verify_node
7
+
8
+ Nothing here is imported by `nabit/__init__.py` — installing nabit never pulls
9
+ in LangGraph/LangChain.
10
+ """
@@ -0,0 +1,176 @@
1
+ """LangGraph / LangChain adapter for nabit.
2
+
3
+ Two ways to plug in, depending on how much you want:
4
+
5
+ 1. NabitCallback — a passive callback handler. Add it once to your graph and it
6
+ verifies every tool result against the checks you register. Zero restructure.
7
+ Detection only (a callback can't re-run a tool), but you still get recording,
8
+ run-id grouping, sinks, and summaries.
9
+
10
+ from nabit.adapters.langgraph import NabitCallback
11
+ from nabit.checks import truthy, has_keys
12
+
13
+ cb = NabitCallback(checks={
14
+ "create_customer": has_keys("id"),
15
+ "search_hotels": truthy(), # catches the empty-list silent fail
16
+ })
17
+ graph.invoke(state, config={"callbacks": [cb]})
18
+ print(cb.summary())
19
+
20
+ 2. verify_node — wrap a single graph node to get the FULL nabit treatment,
21
+ including self-heal retries (a node is just a function, so it can be re-run):
22
+
23
+ from nabit.adapters.langgraph import verify_node
24
+ from nabit.checks import predicate
25
+
26
+ builder.add_node("book", verify_node(
27
+ book_node,
28
+ predicate(lambda update, state: db.has_booking(update["booking_id"])),
29
+ retries=2,
30
+ ))
31
+
32
+ This module lazily imports LangChain — if it isn't installed, `verify_node` and
33
+ the plain decorator still work; only the live callback hookup needs it.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import time
39
+ from typing import Any, Callable, Dict, Optional, Union
40
+
41
+ from ..core import (
42
+ Mode,
43
+ VerificationResult,
44
+ _current_run,
45
+ _emit,
46
+ verify as _verify,
47
+ )
48
+
49
+ try: # lazy / optional — nabit never hard-depends on LangChain
50
+ from langchain_core.callbacks import BaseCallbackHandler
51
+ _HAS_LC = True
52
+ except Exception: # noqa: BLE001
53
+ BaseCallbackHandler = object # type: ignore[assignment,misc]
54
+ _HAS_LC = False
55
+
56
+
57
+ PostCondition = Callable[[Any, dict], bool]
58
+
59
+
60
+ def _extract_output(output: Any) -> Any:
61
+ """LangChain tool output may be a ToolMessage, a str, or a raw value.
62
+ Pull out the payload we should verify."""
63
+ if hasattr(output, "content"):
64
+ return output.content
65
+ return output
66
+
67
+
68
+ class NabitCallback(BaseCallbackHandler):
69
+ """A LangChain/LangGraph callback handler that verifies tool results.
70
+
71
+ Register a `checks` mapping of ``tool_name -> post_condition``. After each
72
+ tool runs, the matching post-condition is checked against the tool's output;
73
+ results are recorded through nabit's normal machinery (so get_results(),
74
+ sinks, and run ids all work). Provide `default_check` to verify every tool.
75
+
76
+ Args:
77
+ checks: {tool_name: (output, ctx) -> bool}. ctx carries {"tool", "inputs"}.
78
+ default_check: applied to any tool without a specific entry.
79
+ mode: Mode.WARN (default), LOG, SILENT, or RAISE. Note RAISE inside a
80
+ callback may be swallowed by the framework — prefer WARN/SILENT and
81
+ inspect summary()/get_results().
82
+ run_id: optional id stamped on every result for grouping.
83
+ """
84
+
85
+ def __init__(
86
+ self,
87
+ checks: Optional[Dict[str, PostCondition]] = None,
88
+ *,
89
+ default_check: Optional[PostCondition] = None,
90
+ mode: Union[Mode, str] = Mode.WARN,
91
+ run_id: Optional[str] = None,
92
+ ) -> None:
93
+ self.checks = checks or {}
94
+ self.default_check = default_check
95
+ self.mode = Mode(mode)
96
+ self.run_id = run_id
97
+ self._names: Dict[str, str] = {}
98
+ self._starts: Dict[str, float] = {}
99
+
100
+ # -- LangChain callback hooks --------------------------------------------
101
+ def on_tool_start(self, serialized, input_str, *, run_id=None, **kwargs): # noqa: ANN001
102
+ name = None
103
+ if isinstance(serialized, dict):
104
+ name = serialized.get("name")
105
+ name = name or kwargs.get("name") or "tool"
106
+ key = str(run_id)
107
+ self._names[key] = name
108
+ self._starts[key] = time.perf_counter()
109
+ # stash inputs for the post-condition context
110
+ self._names[key + ":inputs"] = kwargs.get("inputs") or input_str # type: ignore[assignment]
111
+
112
+ def on_tool_end(self, output, *, run_id=None, **kwargs): # noqa: ANN001
113
+ key = str(run_id)
114
+ name = self._names.pop(key, "tool")
115
+ inputs = self._names.pop(key + ":inputs", None)
116
+ start = self._starts.pop(key, None)
117
+ check = self.checks.get(name, self.default_check)
118
+ if check is None:
119
+ return # nothing registered for this tool — ignore it
120
+ self._run_check(name, check, _extract_output(output), inputs, start)
121
+
122
+ # -- shared logic (framework-independent, so it's unit-testable) ----------
123
+ def _run_check(self, name: str, check: PostCondition, output: Any,
124
+ inputs: Any, start: Optional[float]) -> None:
125
+ ctx = {"tool": name, "inputs": inputs}
126
+ try:
127
+ passed = bool(check(output, ctx))
128
+ err = None
129
+ except Exception as exc: # noqa: BLE001
130
+ passed = False
131
+ err = f"postcondition raised: {exc}"
132
+ dur = (time.perf_counter() - start) * 1000 if start else 0.0
133
+ _emit(
134
+ VerificationResult(
135
+ name=name, passed=passed, duration_ms=dur, result=output,
136
+ error=err, attempts=1,
137
+ run_id=self.run_id or _current_run.get(),
138
+ ),
139
+ self.mode,
140
+ )
141
+
142
+ def summary(self) -> dict:
143
+ """Pass/fail summary scoped to this handler's run_id (if set)."""
144
+ from ..core import summary as _summary
145
+
146
+ return _summary(run_id=self.run_id) if self.run_id else _summary()
147
+
148
+
149
+ def verify_node(
150
+ node: Callable,
151
+ postcondition: PostCondition,
152
+ *,
153
+ mode: Union[Mode, str] = Mode.WARN,
154
+ name: Optional[str] = None,
155
+ retries: int = 0,
156
+ backoff: float = 0.0,
157
+ on_retry: Optional[Callable] = None,
158
+ ):
159
+ """Wrap a LangGraph node function with nabit verification + self-heal.
160
+
161
+ A node is ``(state) -> state_update``. The post-condition receives
162
+ ``(state_update, ctx)`` where ctx holds the bound node args (so ctx["state"]
163
+ is the incoming state). Because a node is a plain function, this path gets
164
+ the full retry/self-heal behavior that the passive callback can't offer.
165
+ """
166
+ return _verify(
167
+ postcondition,
168
+ mode=mode,
169
+ name=name or getattr(node, "__name__", "node"),
170
+ retries=retries,
171
+ backoff=backoff,
172
+ on_retry=on_retry,
173
+ )(node)
174
+
175
+
176
+ __all__ = ["NabitCallback", "verify_node", "_HAS_LC"]
nabit/checks.py ADDED
@@ -0,0 +1,198 @@
1
+ """Composable post-condition builders.
2
+
3
+ Most verifications are the same handful of shapes: "the result has these keys",
4
+ "this file now exists", "this URL returns 2xx", "this row is in the DB". Instead
5
+ of hand-writing a lambda every time (what every other verifier makes you do),
6
+ compose these.
7
+
8
+ Every builder returns a post-condition `(result, context) -> bool` that plugs
9
+ straight into @verify:
10
+
11
+ from nabit import verify
12
+ from nabit.checks import has_keys, all_of, file_exists
13
+
14
+ @verify(all_of(has_keys("id", "status"), lambda r, c: r["status"] == "created"))
15
+ def create(...): ...
16
+
17
+ Zero dependencies. `http_ok` uses only the stdlib (urllib).
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import os
23
+ from typing import Any, Callable, Mapping
24
+
25
+ PostCondition = Callable[[Any, dict], bool]
26
+
27
+
28
+ # --- combinators ------------------------------------------------------------
29
+ def all_of(*checks: PostCondition) -> PostCondition:
30
+ """Pass only if every check passes (logical AND)."""
31
+ def _pc(result: Any, ctx: dict) -> bool:
32
+ return all(bool(c(result, ctx)) for c in checks)
33
+ return _pc
34
+
35
+
36
+ def any_of(*checks: PostCondition) -> PostCondition:
37
+ """Pass if any check passes (logical OR)."""
38
+ def _pc(result: Any, ctx: dict) -> bool:
39
+ return any(bool(c(result, ctx)) for c in checks)
40
+ return _pc
41
+
42
+
43
+ def not_(check: PostCondition) -> PostCondition:
44
+ """Invert a check."""
45
+ def _pc(result: Any, ctx: dict) -> bool:
46
+ return not bool(check(result, ctx))
47
+ return _pc
48
+
49
+
50
+ # --- result-shape checks ----------------------------------------------------
51
+ def has_keys(*keys: str) -> PostCondition:
52
+ """Result is a mapping containing all of `keys`."""
53
+ def _pc(result: Any, ctx: dict) -> bool:
54
+ if not isinstance(result, Mapping):
55
+ return False
56
+ return all(k in result for k in keys)
57
+ return _pc
58
+
59
+
60
+ def equals(expected: Any) -> PostCondition:
61
+ """Result equals `expected`."""
62
+ return lambda result, ctx: result == expected
63
+
64
+
65
+ def field_equals(key: str, expected: Any) -> PostCondition:
66
+ """Result is a mapping where result[key] == expected."""
67
+ def _pc(result: Any, ctx: dict) -> bool:
68
+ return isinstance(result, Mapping) and result.get(key) == expected
69
+ return _pc
70
+
71
+
72
+ def truthy(key: str | None = None) -> PostCondition:
73
+ """Result (or result[key]) is truthy. Catches empty lists/strings/None —
74
+ the classic 'returned [] instead of failing' silent failure."""
75
+ def _pc(result: Any, ctx: dict) -> bool:
76
+ if key is None:
77
+ return bool(result)
78
+ return isinstance(result, Mapping) and bool(result.get(key))
79
+ return _pc
80
+
81
+
82
+ def non_empty() -> PostCondition:
83
+ """Result is non-empty (len > 0). Catches truncated/empty responses."""
84
+ def _pc(result: Any, ctx: dict) -> bool:
85
+ try:
86
+ return len(result) > 0
87
+ except TypeError:
88
+ return result is not None
89
+ return _pc
90
+
91
+
92
+ def _pluck(result: Any, key: Optional[str]) -> Any:
93
+ """Get result[key] if key given and result is a mapping/attr, else result."""
94
+ if key is None:
95
+ return result
96
+ if isinstance(result, Mapping):
97
+ return result.get(key)
98
+ return getattr(result, key, None)
99
+
100
+
101
+ def min_length(n: int, key: Optional[str] = None) -> PostCondition:
102
+ """Result (or result[key]) has len >= n. Catches the 'evidence is just a
103
+ URL' / truncated-output silent failure (e.g. an agent's finding must carry
104
+ >100 chars of real proof, not a one-line claim)."""
105
+ def _pc(result: Any, ctx: dict) -> bool:
106
+ val = _pluck(result, key)
107
+ try:
108
+ return len(val) >= n
109
+ except TypeError:
110
+ return False
111
+ return _pc
112
+
113
+
114
+ def contains(needle: Any, key: Optional[str] = None) -> PostCondition:
115
+ """`needle` is contained in result (or result[key]). Works for substrings,
116
+ list membership, dict keys — anything supporting `in`. Catches 'the proof
117
+ doesn't actually mention the target host' style lies."""
118
+ def _pc(result: Any, ctx: dict) -> bool:
119
+ val = _pluck(result, key)
120
+ try:
121
+ return needle in val
122
+ except TypeError:
123
+ return False
124
+ return _pc
125
+
126
+
127
+ def matches(pattern: str, key: Optional[str] = None) -> PostCondition:
128
+ """A regex search succeeds against result (or result[key]), coerced to str.
129
+ Use for 'evidence must contain a hostname / HTTP status / timestamp'."""
130
+ import re
131
+
132
+ rx = re.compile(pattern)
133
+
134
+ def _pc(result: Any, ctx: dict) -> bool:
135
+ val = _pluck(result, key)
136
+ return val is not None and bool(rx.search(str(val)))
137
+ return _pc
138
+
139
+
140
+ def in_range(lo: float, hi: float, key: Optional[str] = None) -> PostCondition:
141
+ """Numeric result (or result[key]) is within [lo, hi]. Catches values that
142
+ are out of spec (e.g. an LLM judge returning 0s but marking 'passed')."""
143
+ def _pc(result: Any, ctx: dict) -> bool:
144
+ val = _pluck(result, key)
145
+ try:
146
+ return lo <= val <= hi
147
+ except TypeError:
148
+ return False
149
+ return _pc
150
+
151
+
152
+ # --- real-world side-effect checks (the whole point of nabit) ----------------
153
+ def file_exists(path_or_fn: str | Callable[[Any, dict], str]) -> PostCondition:
154
+ """A file exists on disk. `path_or_fn` is a literal path or a callable
155
+ (result, ctx) -> path, so you can derive the path from the agent's output."""
156
+ def _pc(result: Any, ctx: dict) -> bool:
157
+ path = path_or_fn(result, ctx) if callable(path_or_fn) else path_or_fn
158
+ return bool(path) and os.path.exists(path)
159
+ return _pc
160
+
161
+
162
+ def file_fresh(path_or_fn: str | Callable[[Any, dict], str], max_age_s: float) -> PostCondition:
163
+ """A file exists AND was modified within the last `max_age_s` seconds.
164
+ Catches the 'cron job exited 0 but wrote nothing new' failure."""
165
+ import time
166
+
167
+ def _pc(result: Any, ctx: dict) -> bool:
168
+ path = path_or_fn(result, ctx) if callable(path_or_fn) else path_or_fn
169
+ if not path or not os.path.exists(path):
170
+ return False
171
+ return (time.time() - os.path.getmtime(path)) <= max_age_s
172
+ return _pc
173
+
174
+
175
+ def predicate(fn: Callable[[Any, dict], bool]) -> PostCondition:
176
+ """Wrap an arbitrary check against real state — a DB lookup, API call, etc.
177
+ Just sugar for readability / composition:
178
+
179
+ predicate(lambda r, c: db.exists("customers", r["id"]))
180
+ """
181
+ return lambda result, ctx: bool(fn(result, ctx))
182
+
183
+
184
+ def http_ok(url_or_fn: str | Callable[[Any, dict], str], timeout: float = 5.0) -> PostCondition:
185
+ """An HTTP GET to the URL returns a 2xx status. Stdlib only (urllib).
186
+ `url_or_fn` is a literal URL or (result, ctx) -> url."""
187
+ from urllib.request import urlopen
188
+
189
+ def _pc(result: Any, ctx: dict) -> bool:
190
+ url = url_or_fn(result, ctx) if callable(url_or_fn) else url_or_fn
191
+ if not url:
192
+ return False
193
+ try:
194
+ with urlopen(url, timeout=timeout) as resp: # noqa: S310 — caller-supplied URL
195
+ return 200 <= resp.status < 300
196
+ except Exception: # noqa: BLE001
197
+ return False
198
+ return _pc
nabit/core.py ADDED
@@ -0,0 +1,343 @@
1
+ """Core verification primitives for nabit.
2
+
3
+ The whole idea: an LLM agent (or any tool it calls) returns a result that
4
+ *claims* something happened. We don't trust the claim. We run a read-only
5
+ post-condition against the actual system state and record whether reality
6
+ matched the claim.
7
+
8
+ What sets nabit apart from other verifiers:
9
+ * self-heal — on a failed check it can re-run the action (optionally with
10
+ feedback), so it corrects silent failures instead of just reporting them.
11
+ * run-ID correlation — scope a batch of actions under one run so their
12
+ verifications group together (the "nothing shares a run id" problem).
13
+ * pluggable sinks — tee every result to JSONL / OpenTelemetry / your logger
14
+ without nabit taking on a single runtime dependency.
15
+
16
+ Zero dependencies. Works with sync or async functions. Framework-agnostic —
17
+ LangGraph, LangChain, CrewAI, or a plain function all look the same here.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import contextlib
23
+ import contextvars
24
+ import functools
25
+ import inspect
26
+ import logging
27
+ import time
28
+ import uuid
29
+ from dataclasses import asdict, dataclass, field
30
+ from enum import Enum
31
+ from typing import Any, Awaitable, Callable, Optional, Union
32
+
33
+ logger = logging.getLogger("nabit")
34
+
35
+
36
+ class Mode(str, Enum):
37
+ """What to do when a post-condition fails (after retries are exhausted).
38
+
39
+ RAISE — raise VerificationError (fail loud, good for tests / CI).
40
+ WARN — log at WARNING level and return the result anyway (good for prod
41
+ rollout: you get the signal without changing behavior).
42
+ LOG — log at INFO level only.
43
+ SILENT — record the result but emit nothing (inspect via get_results()).
44
+ """
45
+
46
+ RAISE = "raise"
47
+ WARN = "warn"
48
+ LOG = "log"
49
+ SILENT = "silent"
50
+
51
+
52
+ class VerificationError(AssertionError):
53
+ """Raised (in Mode.RAISE) when an agent's claimed outcome does not match
54
+ actual system state, after any retries are exhausted."""
55
+
56
+
57
+ @dataclass
58
+ class VerificationResult:
59
+ """One verification event, kept in an in-memory log for reporting."""
60
+
61
+ name: str
62
+ passed: bool
63
+ duration_ms: float
64
+ result: Any = None
65
+ error: Optional[str] = None
66
+ attempts: int = 1
67
+ run_id: Optional[str] = None
68
+ timestamp: float = field(default_factory=time.time)
69
+
70
+ def to_dict(self) -> dict:
71
+ d = asdict(self)
72
+ # `result` may not be JSON-serializable; stringify defensively.
73
+ try:
74
+ import json
75
+
76
+ json.dumps(d["result"])
77
+ except (TypeError, ValueError):
78
+ d["result"] = repr(d["result"])
79
+ return d
80
+
81
+
82
+ # --- run-ID correlation -----------------------------------------------------
83
+ # A context var so concurrent/async runs don't clobber each other's run id.
84
+ _current_run: contextvars.ContextVar[Optional[str]] = contextvars.ContextVar(
85
+ "nabit_run_id", default=None
86
+ )
87
+
88
+
89
+ @contextlib.contextmanager
90
+ def run(run_id: Optional[str] = None):
91
+ """Scope a batch of verified actions under a single run id, so their
92
+ results group together.
93
+
94
+ with nabit.run() as rid:
95
+ create_customer(...)
96
+ charge_card(...)
97
+ print(nabit.summary(run_id=rid))
98
+
99
+ If `run_id` is omitted a short unique id is generated and yielded.
100
+ """
101
+ rid = run_id or uuid.uuid4().hex[:12]
102
+ token = _current_run.set(rid)
103
+ try:
104
+ yield rid
105
+ finally:
106
+ _current_run.reset(token)
107
+
108
+
109
+ # --- results log + pluggable sinks ------------------------------------------
110
+ _RESULTS: list[VerificationResult] = []
111
+ _MAX_RESULTS = 1000
112
+ _SINKS: list[Callable[[VerificationResult], None]] = []
113
+
114
+
115
+ def add_sink(fn: Callable[[VerificationResult], None]) -> None:
116
+ """Register a callback invoked with every VerificationResult as it happens.
117
+ Use it to tee results to JSONL, OpenTelemetry, Datadog, etc. Zero deps:
118
+ nabit never imports your sink's backend."""
119
+ _SINKS.append(fn)
120
+
121
+
122
+ def clear_sinks() -> None:
123
+ _SINKS.clear()
124
+
125
+
126
+ def get_results(run_id: Optional[str] = None) -> list[VerificationResult]:
127
+ """Return recorded verification results (most recent last). Optionally
128
+ filter to a single run id."""
129
+ if run_id is None:
130
+ return list(_RESULTS)
131
+ return [r for r in _RESULTS if r.run_id == run_id]
132
+
133
+
134
+ def clear_results() -> None:
135
+ """Clear the in-memory results log (useful between tests)."""
136
+ _RESULTS.clear()
137
+
138
+
139
+ def summary(run_id: Optional[str] = None) -> dict:
140
+ """Aggregate pass/fail stats, optionally scoped to one run id."""
141
+ rs = get_results(run_id)
142
+ total = len(rs)
143
+ passed = sum(1 for r in rs if r.passed)
144
+ failed = total - passed
145
+ return {
146
+ "total": total,
147
+ "passed": passed,
148
+ "failed": failed,
149
+ "pass_rate": (passed / total) if total else 1.0,
150
+ "failures": [r.name for r in rs if not r.passed],
151
+ }
152
+
153
+
154
+ def _record(result: VerificationResult) -> None:
155
+ _RESULTS.append(result)
156
+ if len(_RESULTS) > _MAX_RESULTS:
157
+ del _RESULTS[0]
158
+ for sink in _SINKS:
159
+ try:
160
+ sink(result)
161
+ except Exception: # noqa: BLE001 — a broken sink must never break the app
162
+ logger.exception("nabit: sink raised; continuing")
163
+
164
+
165
+ # A post-condition receives (result, context) and returns a truthy value for
166
+ # "reality matches the claim". `context` is the bound call arguments, so you
167
+ # can check the inputs against the outputs. May be sync or async.
168
+ PostCondition = Callable[[Any, dict], Union[bool, Awaitable[bool]]]
169
+ # A retry/feedback hook: (attempt, last_result, context) -> None. Use it to
170
+ # nudge the agent before the next attempt (e.g. append a corrective message).
171
+ RetryHook = Callable[[int, Any, dict], Any]
172
+
173
+
174
+ def _build_context(func: Callable, args: tuple, kwargs: dict) -> dict:
175
+ """Bind call args to parameter names so post-conditions can read inputs."""
176
+ try:
177
+ bound = inspect.signature(func).bind_partial(*args, **kwargs)
178
+ bound.apply_defaults()
179
+ return dict(bound.arguments)
180
+ except TypeError:
181
+ return {"args": args, "kwargs": kwargs}
182
+
183
+
184
+ def _emit(result: VerificationResult, mode: Mode) -> None:
185
+ """Record a final result and raise/log according to mode."""
186
+ _record(result)
187
+ if result.passed:
188
+ logger.debug("nabit: %s verified OK (%.1fms, %d attempt(s))",
189
+ result.name, result.duration_ms, result.attempts)
190
+ return
191
+ msg = (f"nabit: {result.name} FAILED verification after {result.attempts} "
192
+ f"attempt(s) — agent claimed success, reality disagreed")
193
+ if result.error:
194
+ msg += f" ({result.error})"
195
+ if mode is Mode.RAISE:
196
+ raise VerificationError(msg)
197
+ elif mode is Mode.WARN:
198
+ logger.warning(msg)
199
+ elif mode is Mode.LOG:
200
+ logger.info(msg)
201
+ # SILENT: recorded only.
202
+
203
+
204
+ def verify(
205
+ postcondition: PostCondition,
206
+ *,
207
+ mode: Union[Mode, str] = Mode.WARN,
208
+ name: Optional[str] = None,
209
+ on_error: bool = True,
210
+ retries: int = 0,
211
+ backoff: float = 0.0,
212
+ on_retry: Optional[RetryHook] = None,
213
+ ) -> Callable:
214
+ """Decorator: run the wrapped function, then check its claimed outcome
215
+ against actual system state via `postcondition`. On failure, optionally
216
+ self-heal by re-running the action.
217
+
218
+ Args:
219
+ postcondition: callable (result, context) -> bool. Truthy == reality
220
+ matched the claim. May be sync or async.
221
+ mode: what to do once retries are exhausted. See Mode. Default WARN.
222
+ name: label for the result log. Defaults to the function name.
223
+ on_error: if True (default), also run the post-condition when the
224
+ wrapped function raises — so "it threw but the side effect still
225
+ happened" (or vice versa) is handled. On the final failed attempt
226
+ the original exception is re-raised.
227
+ retries: how many times to re-run the action if verification fails
228
+ (0 = no self-heal, just verify once). This is the closed loop:
229
+ detect AND correct.
230
+ backoff: seconds to sleep between retries, multiplied by attempt number
231
+ (linear backoff). Ignored for async if 0.
232
+ on_retry: optional (attempt, last_result, context) -> None hook invoked
233
+ before each retry — use it to feed the discrepancy back to the agent.
234
+
235
+ Works transparently on both sync and async functions.
236
+ """
237
+ mode = Mode(mode)
238
+
239
+ def decorator(func: Callable) -> Callable:
240
+ label = name or getattr(func, "__name__", "anonymous")
241
+ pc_is_async = inspect.iscoroutinefunction(postcondition)
242
+ func_is_async = inspect.iscoroutinefunction(func)
243
+
244
+ async def _run_pc_async(result: Any, ctx: dict) -> bool:
245
+ out = postcondition(result, ctx)
246
+ if inspect.isawaitable(out):
247
+ out = await out
248
+ return bool(out)
249
+
250
+ def _run_pc_sync(result: Any, ctx: dict) -> bool:
251
+ if pc_is_async:
252
+ raise RuntimeError(
253
+ f"nabit: async post-condition used on sync function '{label}'. "
254
+ "Make the wrapped function async, or the post-condition sync."
255
+ )
256
+ return bool(postcondition(result, ctx))
257
+
258
+ def _finalize(result, passed, pc_err, last_exc, attempt, start):
259
+ """Record the final attempt, then raise/return per semantics:
260
+ a real exception always propagates (more informative than a
261
+ VerificationError); otherwise _emit handles raise/warn/log/silent."""
262
+ err = pc_err or (f"function raised: {last_exc}" if last_exc else None)
263
+ vr = VerificationResult(
264
+ name=label, passed=passed,
265
+ duration_ms=(time.perf_counter() - start) * 1000,
266
+ result=result, error=err, attempts=attempt,
267
+ run_id=_current_run.get(),
268
+ )
269
+ if last_exc is not None:
270
+ _record(vr)
271
+ if not passed and mode in (Mode.WARN, Mode.LOG):
272
+ logger.warning("nabit: %s FAILED (function raised after %d "
273
+ "attempt(s)): %s", label, attempt, last_exc)
274
+ raise last_exc
275
+ _emit(vr, mode)
276
+ return result
277
+
278
+ if func_is_async:
279
+ @functools.wraps(func)
280
+ async def async_wrapper(*args, **kwargs):
281
+ import asyncio
282
+
283
+ ctx = _build_context(func, args, kwargs)
284
+ start = time.perf_counter()
285
+ result: Any = None
286
+ for attempt in range(1, retries + 2): # 1 initial + `retries`
287
+ last_exc: Optional[BaseException] = None
288
+ try:
289
+ result = await func(*args, **kwargs)
290
+ except Exception as exc: # noqa: BLE001
291
+ last_exc = exc
292
+ result = None
293
+ if not on_error:
294
+ raise # don't verify, just propagate
295
+ try:
296
+ passed = await _run_pc_async(result, ctx)
297
+ pc_err = None
298
+ except Exception as pc_exc: # noqa: BLE001
299
+ passed = False
300
+ pc_err = f"postcondition raised: {pc_exc}"
301
+ if passed or attempt == retries + 1:
302
+ return _finalize(result, passed, pc_err, last_exc, attempt, start)
303
+ logger.debug("nabit: %s attempt %d failed, retrying", label, attempt)
304
+ if on_retry is not None:
305
+ maybe = on_retry(attempt, result, ctx)
306
+ if inspect.isawaitable(maybe):
307
+ await maybe
308
+ if backoff:
309
+ await asyncio.sleep(backoff * attempt)
310
+
311
+ return async_wrapper
312
+
313
+ @functools.wraps(func)
314
+ def sync_wrapper(*args, **kwargs):
315
+ ctx = _build_context(func, args, kwargs)
316
+ start = time.perf_counter()
317
+ result: Any = None
318
+ for attempt in range(1, retries + 2):
319
+ last_exc: Optional[BaseException] = None
320
+ try:
321
+ result = func(*args, **kwargs)
322
+ except Exception as exc: # noqa: BLE001
323
+ last_exc = exc
324
+ result = None
325
+ if not on_error:
326
+ raise
327
+ try:
328
+ passed = _run_pc_sync(result, ctx)
329
+ pc_err = None
330
+ except Exception as pc_exc: # noqa: BLE001
331
+ passed = False
332
+ pc_err = f"postcondition raised: {pc_exc}"
333
+ if passed or attempt == retries + 1:
334
+ return _finalize(result, passed, pc_err, last_exc, attempt, start)
335
+ logger.debug("nabit: %s attempt %d failed, retrying", label, attempt)
336
+ if on_retry is not None:
337
+ on_retry(attempt, result, ctx)
338
+ if backoff:
339
+ time.sleep(backoff * attempt)
340
+
341
+ return sync_wrapper
342
+
343
+ return decorator
nabit/sinks.py ADDED
@@ -0,0 +1,41 @@
1
+ """Built-in result sinks.
2
+
3
+ A sink is just a callable `(VerificationResult) -> None` registered with
4
+ `nabit.add_sink(...)`. These ship in the box; anything fancier (OpenTelemetry,
5
+ Datadog, Kafka) is a three-line function you write and register — nabit never
6
+ imports those backends itself, so the core stays dependency-free.
7
+
8
+ OpenTelemetry example (you provide the dependency, not nabit):
9
+
10
+ from opentelemetry import trace
11
+ tracer = trace.get_tracer("nabit")
12
+
13
+ def otel_sink(r):
14
+ with tracer.start_as_current_span(f"nabit.verify.{r.name}") as span:
15
+ span.set_attribute("nabit.passed", r.passed)
16
+ span.set_attribute("nabit.attempts", r.attempts)
17
+ if r.run_id:
18
+ span.set_attribute("nabit.run_id", r.run_id)
19
+
20
+ nabit.add_sink(otel_sink)
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import json
26
+ import threading
27
+ from typing import Callable
28
+
29
+
30
+ def jsonl_sink(path: str) -> Callable:
31
+ """Return a sink that appends each VerificationResult as one JSON line to
32
+ `path`. Thread-safe. Register with nabit.add_sink(jsonl_sink("hits.jsonl"))."""
33
+ lock = threading.Lock()
34
+
35
+ def _sink(result) -> None:
36
+ line = json.dumps(result.to_dict(), default=str)
37
+ with lock:
38
+ with open(path, "a", encoding="utf-8") as fh:
39
+ fh.write(line + "\n")
40
+
41
+ return _sink
@@ -0,0 +1,280 @@
1
+ Metadata-Version: 2.5
2
+ Name: nabit
3
+ Version: 0.3.0
4
+ Summary: nab your agent's silent failures — verify LLM-agent outcomes against real system state, with self-heal. Zero dependencies.
5
+ Project-URL: Homepage, https://jakegarnier.com/agent-reliability-kit
6
+ Project-URL: Source, https://github.com/jake-garnier/nabit
7
+ Author-email: Jake Garnier <jakegarnier@gmail.com>
8
+ License: MIT License
9
+
10
+ Copyright (c) 2026 Jake Garnier
11
+
12
+ Permission is hereby granted, free of charge, to any person obtaining a copy
13
+ of this software and associated documentation files (the "Software"), to deal
14
+ in the Software without restriction, including without limitation the rights
15
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
16
+ copies of the Software, and to permit persons to whom the Software is
17
+ furnished to do so, subject to the following conditions:
18
+
19
+ The above copyright notice and this permission notice shall be included in all
20
+ copies or substantial portions of the Software.
21
+
22
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
23
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
24
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
25
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
26
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
27
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
28
+ SOFTWARE.
29
+ License-File: LICENSE
30
+ Keywords: agents,ai,langchain,langgraph,llm,observability,reliability,verification
31
+ Classifier: Development Status :: 4 - Beta
32
+ Classifier: Intended Audience :: Developers
33
+ Classifier: License :: OSI Approved :: MIT License
34
+ Classifier: Programming Language :: Python :: 3
35
+ Classifier: Topic :: Software Development :: Quality Assurance
36
+ Requires-Python: >=3.9
37
+ Provides-Extra: dev
38
+ Requires-Dist: langchain-core>=0.2; extra == 'dev'
39
+ Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
40
+ Requires-Dist: pytest>=7; extra == 'dev'
41
+ Provides-Extra: langgraph
42
+ Requires-Dist: langchain-core>=0.2; extra == 'langgraph'
43
+ Description-Content-Type: text/markdown
44
+
45
+ # nabit — *nab your agent's silent failures*
46
+
47
+ Your LLM agent said it created the customer. Your database says otherwise. You
48
+ found out three days later from a support ticket.
49
+
50
+ **nabit** is a dead-simple verification layer for LLM agents. Your agent claims
51
+ it's done; `nabit` checks the *real system state*, tells you whether that was
52
+ true, and — if it wasn't — **re-runs the action to fix it.** One decorator.
53
+ Zero dependencies. Sync or async.
54
+
55
+ ```python
56
+ from nabit import verify
57
+
58
+ @verify(lambda result, ctx: db.exists("customers", result["id"]),
59
+ retries=2) # self-heal: re-run on failure
60
+ def create_customer(name):
61
+ # agent / tool does the work and claims success
62
+ return {"id": 42, "status": "created"}
63
+ ```
64
+
65
+ If the agent returns `{"status": "created"}` but the row isn't in the database,
66
+ `nabit` catches the lie instead of letting a green dashboard hide it — then
67
+ retries the action up to `retries` times before giving up.
68
+
69
+ ## How it's different
70
+
71
+ Most tools in this space either detect problems without fixing them, only check
72
+ the *text the model produced* (not whether the real action happened), or make you
73
+ stand up a backend to do it. nabit:
74
+
75
+ - **checks real side effects**, not output schema (vs Guardrails AI / Instructor)
76
+ - **closes the loop** — self-heal retries, not just detection (vs Drift / trace viewers)
77
+ - **verifies inline at runtime**, not in a postmortem (vs agent-coroner)
78
+ - **has zero dependencies and no backend** — it's a decorator, not a platform (vs COGEXT)
79
+
80
+ ## Why this exists
81
+
82
+ Trace viewers (Langfuse, LangSmith, Helicone) answer *"what did the agent do?"*
83
+ really well. They don't answer *"was what it did actually correct?"*
84
+
85
+ The expensive failures are **semantic**, not technical: no exception thrown, the
86
+ tool call succeeded, and the output was still wrong. Those look identical to a
87
+ success at the log level. `nabit` is the layer that checks the claim against
88
+ reality.
89
+
90
+ ## Install
91
+
92
+ ```bash
93
+ pip install nabit
94
+ ```
95
+
96
+ ## Usage
97
+
98
+ ### Post-conditions check reality, not the agent's word
99
+
100
+ A post-condition is `(result, context) -> bool`. `result` is what the function
101
+ returned; `context` is the bound call arguments (so you can compare inputs to
102
+ outputs). Return truthy if reality matches the claim.
103
+
104
+ ```python
105
+ from nabit import verify
106
+
107
+ @verify(lambda result, ctx: ticket_store.is_closed(result["ticket_id"]))
108
+ def close_ticket(ticket_id):
109
+ agent.act(f"close ticket {ticket_id}")
110
+ return {"ticket_id": ticket_id, "status": "closed"}
111
+ ```
112
+
113
+ ### Modes — tune how loud failures are
114
+
115
+ ```python
116
+ from nabit import verify, Mode
117
+
118
+ @verify(check, mode=Mode.WARN) # log a warning, keep running (default — safe for prod)
119
+ @verify(check, mode=Mode.RAISE) # raise VerificationError (great for tests / CI)
120
+ @verify(check, mode=Mode.LOG) # info-level log only
121
+ @verify(check, mode=Mode.SILENT) # record only; inspect later
122
+ ```
123
+
124
+ ### Self-heal — re-run the action when verification fails
125
+
126
+ ```python
127
+ def feedback(attempt, last_result, ctx):
128
+ # optional: nudge the agent before the next attempt
129
+ log.warning("verification failed on attempt %d, retrying", attempt)
130
+
131
+ @verify(check, retries=3, backoff=0.5, on_retry=feedback)
132
+ def book_flight(req):
133
+ ...
134
+ ```
135
+
136
+ `retries` re-runs the whole action up to N times until the post-condition
137
+ passes; `backoff` adds linear delay between attempts; `on_retry` lets you feed
138
+ the discrepancy back to the agent. Detection *and* correction, in one decorator.
139
+
140
+ ### Composable checks (no hand-written lambdas)
141
+
142
+ ```python
143
+ from nabit import verify
144
+ from nabit.checks import (all_of, has_keys, field_equals, file_fresh, http_ok,
145
+ truthy, min_length, contains, matches, in_range)
146
+
147
+ @verify(all_of(has_keys("id", "status"), field_equals("status", "created")))
148
+ def create(...): ...
149
+
150
+ @verify(file_fresh(lambda r, c: r["path"], max_age_s=60)) # cron wrote a fresh file?
151
+ def nightly_report(): ...
152
+
153
+ @verify(http_ok(lambda r, c: r["url"])) # deployed URL is live?
154
+ def deploy(): ...
155
+
156
+ @verify(truthy()) # not [] / "" / None
157
+ def search(...): ...
158
+
159
+ # an agent's "reportable" finding must carry real proof, not a one-line claim
160
+ @verify(all_of(min_length(100, key="evidence"), matches(r"https?://", key="evidence")))
161
+ def finish_task(finding): ...
162
+
163
+ @verify(in_range(1, 10, key="score")) # LLM judge score in spec
164
+ def judge(...): ...
165
+ ```
166
+
167
+ Full check list: `all_of` / `any_of` / `not_`, `has_keys`, `equals`,
168
+ `field_equals`, `truthy`, `non_empty`, `min_length`, `contains`, `matches`,
169
+ `in_range`, `file_exists`, `file_fresh`, `http_ok`, `predicate`.
170
+
171
+ ### Group verifications under a run id
172
+
173
+ ```python
174
+ from nabit import run, summary
175
+
176
+ with run("signup-flow") as rid:
177
+ create_customer(...)
178
+ charge_card(...)
179
+
180
+ print(summary(run_id=rid))
181
+ # {'total': 2, 'passed': 1, 'failed': 1, 'pass_rate': 0.5, 'failures': ['charge_card']}
182
+ ```
183
+
184
+ Solves the "nothing shares a run id" problem — all the checks for one logical
185
+ task carry the same id.
186
+
187
+ ### Tee results anywhere (pluggable sinks, still zero-dep)
188
+
189
+ ```python
190
+ import nabit
191
+ from nabit import jsonl_sink
192
+
193
+ nabit.add_sink(jsonl_sink("verifications.jsonl")) # built-in
194
+
195
+ def otel_sink(r): # or your own, 3 lines
196
+ span.set_attribute("nabit.passed", r.passed)
197
+ nabit.add_sink(otel_sink)
198
+ ```
199
+
200
+ ### LangGraph / LangChain
201
+
202
+ nabit works with any framework via the decorator, but LangGraph users get a
203
+ first-class adapter (lazily imported — installing nabit never pulls in
204
+ LangChain). Two options:
205
+
206
+ **Passive callback** — add it once, verify every tool result, no restructure:
207
+
208
+ ```python
209
+ from nabit.adapters.langgraph import NabitCallback
210
+ from nabit.checks import has_keys, truthy
211
+
212
+ cb = NabitCallback(checks={
213
+ "create_customer": has_keys("id"),
214
+ "search_hotels": truthy(), # catches the empty-list silent failure
215
+ })
216
+ graph.invoke(state, config={"callbacks": [cb]})
217
+ print(cb.summary())
218
+ ```
219
+
220
+ **Node wrapper** — wrap a node to get full self-heal (a node is just a function,
221
+ so it can be re-run):
222
+
223
+ ```python
224
+ from nabit.adapters.langgraph import verify_node
225
+ from nabit.checks import predicate
226
+
227
+ builder.add_node("book", verify_node(
228
+ book_node,
229
+ predicate(lambda update, ctx: db.has_booking(update["booking_id"])),
230
+ retries=2,
231
+ ))
232
+ ```
233
+
234
+ ### Async works the same way
235
+
236
+ ```python
237
+ @verify(pc, retries=2) # async funcs + async post-conditions
238
+ async def create_order(cart):
239
+ ...
240
+ ```
241
+
242
+ ### Inspect what happened
243
+
244
+ ```python
245
+ from nabit import get_results
246
+
247
+ for r in get_results():
248
+ print(r.name, "PASS" if r.passed else "FAIL",
249
+ f"{r.duration_ms:.0f}ms", f"attempts={r.attempts}", r.error)
250
+ ```
251
+
252
+ ### Also catches "it threw but the side effect still happened"
253
+
254
+ By default (`on_error=True`) the post-condition runs even when the wrapped
255
+ function raises — so a tool that errors *after* mutating state (or succeeds in
256
+ reality despite throwing) still gets verified. The original exception is
257
+ re-raised after recording.
258
+
259
+ ## Scope
260
+
261
+ `nabit` is the **outcome verifier**: it answers *"did this specific action
262
+ actually happen?"* and corrects it when it didn't. It deliberately does **not**
263
+ try to be an observability platform. For *fleet-wide* behavioral monitoring —
264
+ real-time degradation detection across many runs (step-count blowups, token
265
+ spikes, slow drift over hundreds of runs), trajectory snapshot diffing, and a
266
+ dashboard — see **[The Production Agent Reliability Kit](https://jakegarnier.com/agent-reliability-kit)**,
267
+ built on published research
268
+ ([SENTINEL](https://github.com/jake-garnier/sentinel), self-supervised anomaly
269
+ detection for LLM agents).
270
+
271
+ Rule of thumb: use `nabit` to verify *one action's* real effect inline; reach
272
+ for the Kit when you need to watch *patterns across runs* over time.
273
+
274
+ ## License
275
+
276
+ MIT — see [LICENSE](LICENSE). Use it anywhere, including commercially.
277
+
278
+ ---
279
+
280
+ Built by [Jake Garnier](https://jakegarnier.com).
@@ -0,0 +1,10 @@
1
+ nabit/__init__.py,sha256=D8xszU7YpoGstTrqhW6ZsqXBJSWplEgESqFpA1gDvG0,1458
2
+ nabit/checks.py,sha256=2T2ITrojCd0NMsbonlXt7qdDhooSp6zEN4PbW33EU40,7070
3
+ nabit/core.py,sha256=sC5ClyIUH28L7JVFjjR-S2zyTSt6eQOYy0Ag-N-AjUA,13070
4
+ nabit/sinks.py,sha256=qMTdUtLf7-iuRkwL7AZziUNgF0m5TcBehL9vc1m4HQk,1351
5
+ nabit/adapters/__init__.py,sha256=KvQWOT8T84zKPC6a2E992uJeQ6cYyuNZrd62RRmDFAE,369
6
+ nabit/adapters/langgraph.py,sha256=2yD7DqEp0vG0Ji_3_HWae0rRZf6uRWqLSOdLyHnZlPQ,6483
7
+ nabit-0.3.0.dist-info/METADATA,sha256=zne-jGZ_7CjBECrw6THfE4QbnxXW5QDbD81xPnuGOuQ,10220
8
+ nabit-0.3.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
9
+ nabit-0.3.0.dist-info/licenses/LICENSE,sha256=P-wcQkV8adB4cgZ4km_ys9Opnln6bRucGHHIcR_N8cM,1069
10
+ nabit-0.3.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Jake Garnier
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.