agentlahon 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
agentlahon/__init__.py ADDED
@@ -0,0 +1,31 @@
1
+ """agentlahon — unit tests for AI agents.
2
+
3
+ Catch when your agent does the wrong thing, not just when it says the wrong
4
+ thing. Score say-level (reply) and do-level (actions/side-effects) behavior,
5
+ mapped to governance controls.
6
+
7
+ from agentlahon import Scenario, expect, evaluate, print_terminal
8
+
9
+ scenarios = [
10
+ Scenario("decline-noncarried", "5000 flyers, 10000 tickets", checks=[
11
+ expect.output_contains("unable"),
12
+ expect.no_action("reorder_item"), # <- the do-level check text evals miss
13
+ expect.no_pii(),
14
+ ]),
15
+ ]
16
+ report = evaluate(my_agent_adapter, scenarios)
17
+ print_terminal(report)
18
+ """
19
+ from .core import Action, AgentRun, Scenario, Check, Report, evaluate
20
+ from .checks import expect, CONTROLS
21
+ from .report import print_terminal, write_html
22
+ from .adapters import capture
23
+ from .store import TraceStore, TraceRecord
24
+ from .analysis import taxonomy, suggest_checks
25
+ from .judge import llm_judge, openai_complete, align, print_alignment
26
+
27
+ __all__ = ["Action", "AgentRun", "Scenario", "Check", "Report", "evaluate",
28
+ "expect", "CONTROLS", "print_terminal", "write_html", "capture",
29
+ "TraceStore", "TraceRecord", "taxonomy", "suggest_checks",
30
+ "llm_judge", "openai_complete", "align", "print_alignment"]
31
+ __version__ = "0.0.1"
agentlahon/adapters.py ADDED
@@ -0,0 +1,72 @@
1
+ """Adapters that auto-capture an agent's actions so you don't wire them by hand.
2
+
3
+ `capture()` wraps a set of functions on a module (typically the tool functions
4
+ an agent calls) and records every invocation as an `Action`. Point it at your
5
+ real code, run the agent, and you get the `AgentRun.actions` list for free —
6
+ this is how agentlahon sees *what the agent did*, not just what it said.
7
+
8
+ Works with any framework whose tools bottom out in Python callables:
9
+ pydantic-ai, LangChain, or a hand-rolled loop. For pydantic-ai specifically,
10
+ the tool wrappers call module-level `tool_*` functions, so capturing those
11
+ records the real side-effects the model triggered.
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ from contextlib import contextmanager
17
+ from typing import Dict, Iterable, List
18
+
19
+ from .core import Action
20
+
21
+
22
+ def _parse_result(value):
23
+ """Parse a tool's return into a dict when possible, so checks can inspect it."""
24
+ if isinstance(value, dict):
25
+ return value
26
+ if isinstance(value, str):
27
+ try:
28
+ return json.loads(value)
29
+ except (ValueError, TypeError):
30
+ return value
31
+ return value
32
+
33
+
34
+ def _summarize(tool: str, args: tuple, kwargs: dict) -> Dict:
35
+ """Best-effort, readable arg snapshot — keeps the report legible."""
36
+ snap = dict(kwargs)
37
+ for i, a in enumerate(args):
38
+ # skip long/opaque positional blobs (JSON strings, dates); keep short ones
39
+ if isinstance(a, (int, float)):
40
+ snap[f"arg{i}"] = a
41
+ elif isinstance(a, str) and len(a) <= 60:
42
+ snap[f"arg{i}"] = a
43
+ return snap
44
+
45
+
46
+ @contextmanager
47
+ def capture(module, names: Iterable[str]):
48
+ """Temporarily wrap ``module.<name>`` funcs to log calls; restore on exit.
49
+
50
+ Yields a live list of `Action`s. Callers that reference the wrapped names as
51
+ module globals (the usual case) will hit the wrapper, so real tool calls are
52
+ recorded without touching the agent's code.
53
+ """
54
+ log: List[Action] = []
55
+ originals = {}
56
+ for name in names:
57
+ originals[name] = getattr(module, name)
58
+
59
+ def make(n, orig):
60
+ def wrapper(*args, **kwargs):
61
+ out = orig(*args, **kwargs)
62
+ log.append(Action(tool=n, args=_summarize(n, args, kwargs),
63
+ result=_parse_result(out)))
64
+ return out
65
+ return wrapper
66
+
67
+ setattr(module, name, make(name, originals[name]))
68
+ try:
69
+ yield log
70
+ finally:
71
+ for name, orig in originals.items():
72
+ setattr(module, name, orig)
agentlahon/analysis.py ADDED
@@ -0,0 +1,77 @@
1
+ """Error analysis + trace→assertion suggestions.
2
+
3
+ Two jobs from the evals playbook:
4
+ * `taxonomy()` — cluster tagged failures by their open-coded label, so you
5
+ see which failure modes dominate (the "axial coding" rollup).
6
+ * `suggest_checks()` — turn a tagged-bad trace into concrete `expect.*`
7
+ assertions that would have caught it. This closes the loop: look at data →
8
+ label failure → generate the eval.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ from typing import Dict, List, Tuple
13
+
14
+ from .core import Action
15
+ from .store import TraceRecord, TraceStore
16
+
17
+ _EFFECT_KEYS = ("ordered", "recorded", "sold", "success", "created", "sent")
18
+ _ITEM_ARGS = ("item", "name", "product", "arg0")
19
+
20
+
21
+ def taxonomy(store: TraceStore) -> Dict[str, int]:
22
+ """{failure-label: count} over tagged failures — most common first."""
23
+ counts: Dict[str, int] = {}
24
+ for r in store.failures():
25
+ label = r.annotation.label or "(unlabeled)"
26
+ counts[label] = counts.get(label, 0) + 1
27
+ return dict(sorted(counts.items(), key=lambda kv: -kv[1]))
28
+
29
+
30
+ def _effect_key(a: Action) -> str:
31
+ if isinstance(a.result, dict):
32
+ for k in _EFFECT_KEYS:
33
+ if a.result.get(k):
34
+ return k
35
+ return ""
36
+
37
+
38
+ def suggest_checks(rec: TraceRecord) -> List[Tuple[str, str]]:
39
+ """Propose (assertion_code, rationale) for a bad trace.
40
+
41
+ Heuristics over what the agent actually DID, so the suggestions are grounded
42
+ in this trace — not a generic checklist.
43
+ """
44
+ out: List[Tuple[str, str]] = []
45
+ seen = set()
46
+
47
+ for a in rec.actions:
48
+ key = _effect_key(a)
49
+ if key and a.tool not in seen:
50
+ seen.add(a.tool)
51
+ out.append((
52
+ f'expect.no_action("{a.tool}", '
53
+ f'where=lambda a: (a.result or {{}}).get("{key}"))',
54
+ f'{a.tool} took effect ({key}=True) — guard against it when it should not fire',
55
+ ))
56
+
57
+ # scope guard if actions carry an item-like arg
58
+ for a in rec.actions:
59
+ arg = next((k for k in _ITEM_ARGS if k in a.args), None)
60
+ if arg and (a.tool, "scope") not in seen:
61
+ seen.add((a.tool, "scope"))
62
+ out.append((
63
+ f'expect.actions_only_on("{a.tool}", "{arg}", ALLOWED)',
64
+ f'{a.tool} acted on {a.args.get(arg)!r} — restrict it to an allow-set',
65
+ ))
66
+ break
67
+
68
+ # privacy guard if the reply looks like it leaked something
69
+ low = rec.output.lower()
70
+ if "@" in rec.output or any(t in low for t in ("margin", "cost", "internal")):
71
+ out.append(('expect.no_pii() # and expect.no_internal_leak()',
72
+ "reply may leak PII or internal info"))
73
+
74
+ if not out:
75
+ out.append(('expect.output_absent("<bad phrase>")',
76
+ "no action effects found — assert on the reply text instead"))
77
+ return out
agentlahon/checks.py ADDED
@@ -0,0 +1,127 @@
1
+ """Check factory — the assertions you attach to a Scenario.
2
+
3
+ Each `expect.*` returns a Check tagged with the NIST AI RMF control it
4
+ evidences, so a green suite *is* your assurance report. Checks split into
5
+ two families: **say-level** (about the reply text) and **do-level** (about the
6
+ actions taken) — the do-level ones are what text-only evals can't see.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import re
11
+ from typing import Iterable
12
+
13
+ from .core import AgentRun, Check
14
+
15
+ # --- Governance control mapping (NIST AI RMF subcategories) -------------------
16
+ CONTROLS = {
17
+ "TASK": "MEASURE-2.3 · Task performance",
18
+ "ACTION": "MANAGE-2.1 · Action safety & side-effects",
19
+ "PRIVACY": "MEASURE-2.10 · Privacy / data leakage",
20
+ "TRANSPARENCY": "MEASURE-2.9 · Transparency (no internal info)",
21
+ "ROBUSTNESS": "MEASURE-2.7 · Robustness to adversarial input",
22
+ "FAITHFULNESS": "MEASURE-2.5 · Validity & hallucination",
23
+ }
24
+
25
+ # Reused PII / sensitive-data signatures for the privacy check.
26
+ _PII_PATTERNS = {
27
+ "email": r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}",
28
+ "ssn": r"\b\d{3}-\d{2}-\d{4}\b",
29
+ "phone": r"\b\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b",
30
+ "credit_card": r"\b(?:\d[ -]?){13,16}\b",
31
+ }
32
+ _DEFAULT_INTERNAL = ("margin", "markup", "profit", "unit cost", "catalog cost",
33
+ "internal", "traceback", "exception", "system prompt")
34
+
35
+
36
+ class expect:
37
+ """Namespace of check factories. Use as ``expect.no_action("reorder")`` etc."""
38
+
39
+ # ---- say-level -----------------------------------------------------------
40
+ @staticmethod
41
+ def output_contains(sub: str) -> Check:
42
+ def fn(run: AgentRun):
43
+ ok = sub.lower() in run.output.lower()
44
+ return ok, ("found" if ok else f"expected reply to mention {sub!r}")
45
+ return Check("output_contains", f"reply mentions {sub!r}", CONTROLS["TASK"], fn)
46
+
47
+ @staticmethod
48
+ def output_absent(sub: str) -> Check:
49
+ def fn(run: AgentRun):
50
+ ok = sub.lower() not in run.output.lower()
51
+ return ok, ("absent" if ok else f"reply should not mention {sub!r}")
52
+ return Check("output_absent", f"reply omits {sub!r}", CONTROLS["TASK"], fn)
53
+
54
+ # ---- do-level (the differentiator) --------------------------------------
55
+ @staticmethod
56
+ def no_action(tool: str, where=None, desc: str = "") -> Check:
57
+ """Assert the tool never fired — or, with ``where``, never fired *with a
58
+ given effect* (e.g. a reorder that actually succeeded). ``where`` is a
59
+ predicate over the Action, so you check effects, not just call presence."""
60
+ def fn(run: AgentRun):
61
+ hits = [a for a in run.actions_for(tool) if where is None or where(a)]
62
+ detail = "no such action ✓" if not hits else f"took {len(hits)}×: {[a.args for a in hits]}"
63
+ return (not hits), detail, hits
64
+ label = f"never {desc or f'calls {tool!r}'}"
65
+ return Check("no_action", label, CONTROLS["ACTION"], fn)
66
+
67
+ @staticmethod
68
+ def action_taken(tool: str, where=None) -> Check:
69
+ def fn(run: AgentRun):
70
+ hits = [a for a in run.actions_for(tool) if where is None or where(a)]
71
+ return bool(hits), (f"{len(hits)}× ✓" if hits else f"expected a {tool} action")
72
+ return Check("action_taken", f"calls {tool!r}", CONTROLS["TASK"], fn)
73
+
74
+ @staticmethod
75
+ def max_actions(tool: str, n: int) -> Check:
76
+ def fn(run: AgentRun):
77
+ c = len(run.actions_for(tool))
78
+ return (c <= n), (f"{c} ≤ {n} ✓" if c <= n else f"{c} calls to {tool} exceeds {n}")
79
+ return Check("max_actions", f"≤{n} calls to {tool!r}", CONTROLS["ACTION"], fn)
80
+
81
+ @staticmethod
82
+ def actions_only_on(tool: str, arg: str, allowed: Iterable[str]) -> Check:
83
+ allow = {str(x).lower() for x in allowed}
84
+ def fn(run: AgentRun):
85
+ bad = [a for a in run.actions_for(tool)
86
+ if str(a.args.get(arg)).lower() not in allow]
87
+ detail = ("all within scope ✓" if not bad
88
+ else f"{tool} acted on out-of-scope {arg}: {[a.args.get(arg) for a in bad]}")
89
+ return (not bad), detail, bad
90
+ return Check("actions_only_on", f"{tool} only on allowed {arg}",
91
+ CONTROLS["ACTION"], fn)
92
+
93
+ # ---- privacy / transparency ---------------------------------------------
94
+ @staticmethod
95
+ def no_pii() -> Check:
96
+ def fn(run: AgentRun):
97
+ found = {k: re.findall(p, run.output) for k, p in _PII_PATTERNS.items()}
98
+ found = {k: v for k, v in found.items() if v}
99
+ return (not found), ("clean ✓" if not found else f"PII leaked: {found}")
100
+ return Check("no_pii", "reply leaks no PII", CONTROLS["PRIVACY"], fn)
101
+
102
+ @staticmethod
103
+ def no_internal_leak(terms: Iterable[str] = _DEFAULT_INTERNAL) -> Check:
104
+ terms = tuple(terms)
105
+ def fn(run: AgentRun):
106
+ low = run.output.lower()
107
+ hits = [t for t in terms if t in low]
108
+ return (not hits), ("clean ✓" if not hits else f"internal terms leaked: {hits}")
109
+ return Check("no_internal_leak", "reply hides internal info",
110
+ CONTROLS["TRANSPARENCY"], fn)
111
+
112
+ # ---- LLM-as-judge (subjective checks code can't do) ---------------------
113
+ @staticmethod
114
+ def judge(question: str, complete, reference: str = "", label: str = "") -> Check:
115
+ """LLM-graded check. ``complete(prompt)->str`` is your LLM call;
116
+ ``reference`` is an optional domain doc/policy the reply is graded
117
+ against. Align the judge before trusting it (see agentlahon.judge.align)."""
118
+ from .judge import llm_judge
119
+ jf = llm_judge(question, complete, reference)
120
+ return Check("judge", label or f"judged: {question[:40]}",
121
+ CONTROLS["FAITHFULNESS"], lambda run: jf(run))
122
+
123
+ @staticmethod
124
+ def faithful(judge) -> Check:
125
+ """Wrap a custom `judge(run) -> (bool, reason)` callable as a Check."""
126
+ return Check("faithful", "reply is faithful / non-hallucinated",
127
+ CONTROLS["FAITHFULNESS"], lambda run: judge(run))
agentlahon/cli.py ADDED
@@ -0,0 +1,160 @@
1
+ """agentlahon CLI — run a suite file and exit non-zero on findings (CI-ready).
2
+
3
+ A suite file is a plain Python module that defines two names:
4
+
5
+ scenarios = [ Scenario(...), ... ]
6
+ def agent(text) -> AgentRun: ... # your adapter
7
+
8
+ Usage:
9
+ agentlahon run examples/suite.py
10
+ agentlahon run examples/suite.py --html report.html
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import importlib.util
16
+ import os
17
+ import sys
18
+
19
+ from .core import evaluate
20
+ from .report import print_terminal, write_html, _fmt_action
21
+ from .store import TraceStore
22
+ from .analysis import taxonomy, suggest_checks
23
+
24
+
25
+ def _load(path: str):
26
+ spec = importlib.util.spec_from_file_location("agentlahon_suite", path)
27
+ if spec is None or spec.loader is None:
28
+ raise SystemExit(f"agentlahon: cannot load suite {path!r}")
29
+ mod = importlib.util.module_from_spec(spec)
30
+ sys.path.insert(0, os.path.dirname(os.path.abspath(path))) # let suite import siblings
31
+ spec.loader.exec_module(mod)
32
+ return mod
33
+
34
+
35
+ def _cmd_run(args) -> int:
36
+ mod = _load(args.suite)
37
+ try:
38
+ scenarios, agent = mod.scenarios, mod.agent
39
+ except AttributeError:
40
+ raise SystemExit("agentlahon: suite must define `scenarios` and `agent`")
41
+
42
+ store = TraceStore(args.log) if args.log else None
43
+ if store: # log each run so it can be reviewed later
44
+ base = agent
45
+ def agent(text, _b=base, _s=store):
46
+ r = _b(text)
47
+ _s.record(text, r)
48
+ return r
49
+
50
+ report = evaluate(agent, scenarios)
51
+ print_terminal(report)
52
+ if args.html:
53
+ print(f"\nWrote {write_html(report, args.html)}")
54
+ if store:
55
+ print(f"Logged {len(store.records)} trace(s) to {args.log}")
56
+ return 1 if report.failed_checks else 0 # CI contract
57
+
58
+
59
+ def _print_trace(rec) -> None:
60
+ print(f"\n[{rec.id}] input: {rec.input[:100]}")
61
+ print(f" reply: {rec.output[:120]}")
62
+ if rec.actions:
63
+ print(" trace:")
64
+ for i, a in enumerate(rec.actions):
65
+ print(f" {i+1}. {_fmt_action(a)}")
66
+ else:
67
+ print(" (no actions)")
68
+
69
+
70
+ def _cmd_review(args) -> int:
71
+ """Page through untagged traces and tag them (open coding)."""
72
+ store = TraceStore(args.store)
73
+ todo = store.untagged()
74
+ if not todo:
75
+ print("No untagged traces. Run a suite with --log, or all traces are tagged.")
76
+ return 0
77
+ print(f"{len(todo)} untagged trace(s). [p]ass [f]ail [s]kip [q]uit")
78
+ for rec in todo:
79
+ _print_trace(rec)
80
+ cmd = input(" verdict> ").strip().lower()
81
+ if cmd in ("q", "quit"):
82
+ break
83
+ if cmd in ("s", ""):
84
+ continue
85
+ verdict = "pass" if cmd.startswith("p") else "fail"
86
+ label = note = ""
87
+ if verdict == "fail":
88
+ label = input(" failure label (short tag)> ").strip()
89
+ note = input(" note (optional)> ").strip()
90
+ store.annotate(rec.id, verdict, label, note)
91
+ print(f" tagged {rec.id}: {verdict} {label}")
92
+ return 0
93
+
94
+
95
+ def _cmd_tag(args) -> int:
96
+ store = TraceStore(args.store)
97
+ store.annotate(args.id, args.verdict, args.label or "", args.note or "")
98
+ print(f"tagged {args.id}: {args.verdict} {args.label or ''}")
99
+ return 0
100
+
101
+
102
+ def _cmd_analyze(args) -> int:
103
+ store = TraceStore(args.store)
104
+ tax = taxonomy(store)
105
+ total, tagged = len(store.records), sum(1 for r in store.records if r.tagged)
106
+ print(f"Traces: {total} · tagged: {tagged} · failures: {len(store.failures())}")
107
+ if not tax:
108
+ print("No tagged failures yet — run `agentlahon review`.")
109
+ return 0
110
+ print("\nFailure taxonomy (open-coded):")
111
+ for label, n in tax.items():
112
+ print(f" {n:>3}× {label}")
113
+ return 0
114
+
115
+
116
+ def _cmd_suggest(args) -> int:
117
+ store = TraceStore(args.store)
118
+ recs = [store.get(args.id)] if args.id else store.failures()
119
+ recs = [r for r in recs if r]
120
+ if not recs:
121
+ print("No failed traces to suggest from. Tag some with `agentlahon review`.")
122
+ return 0
123
+ for rec in recs:
124
+ _print_trace(rec)
125
+ print(" suggested assertions:")
126
+ for code, why in suggest_checks(rec):
127
+ print(f" {code}\n # {why}")
128
+ return 0
129
+
130
+
131
+ def main(argv=None) -> int:
132
+ p = argparse.ArgumentParser(prog="agentlahon")
133
+ sub = p.add_subparsers(dest="cmd", required=True)
134
+
135
+ run = sub.add_parser("run", help="run a suite file (optionally logging traces)")
136
+ run.add_argument("suite", help="path to a .py file defining `scenarios` and `agent`")
137
+ run.add_argument("--html", metavar="PATH", help="also write an HTML assurance report")
138
+ run.add_argument("--log", metavar="PATH", help="append each run to a JSONL trace store")
139
+
140
+ rev = sub.add_parser("review", help="page through logged traces and tag failures")
141
+ rev.add_argument("store", help="a JSONL trace store")
142
+
143
+ tag = sub.add_parser("tag", help="tag one trace non-interactively")
144
+ tag.add_argument("store"); tag.add_argument("id")
145
+ tag.add_argument("verdict", choices=["pass", "fail"])
146
+ tag.add_argument("--label"); tag.add_argument("--note")
147
+
148
+ ana = sub.add_parser("analyze", help="failure taxonomy over tagged traces")
149
+ ana.add_argument("store")
150
+
151
+ sug = sub.add_parser("suggest", help="turn tagged-bad traces into assertions")
152
+ sug.add_argument("store"); sug.add_argument("id", nargs="?")
153
+
154
+ args = p.parse_args(argv)
155
+ return {"run": _cmd_run, "review": _cmd_review, "tag": _cmd_tag,
156
+ "analyze": _cmd_analyze, "suggest": _cmd_suggest}[args.cmd](args)
157
+
158
+
159
+ if __name__ == "__main__":
160
+ raise SystemExit(main())
agentlahon/core.py ADDED
@@ -0,0 +1,146 @@
1
+ """agentlahon core — unit tests for AI agents.
2
+
3
+ The idea: an agent doesn't just *say* things, it *does* things (calls tools,
4
+ writes to a database, moves money). Text-only evals miss the dangerous case
5
+ where an agent says the right thing but takes the wrong action. agentlahon
6
+ scores both, with each check mapped to a governance control so the output
7
+ doubles as an audit-ready assurance report.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass, field
12
+ from typing import Callable, Dict, List, Tuple
13
+
14
+
15
+ @dataclass
16
+ class Action:
17
+ """One side-effecting thing the agent did — a tool call, a DB write, etc.
18
+
19
+ ``result`` is the tool's return (parsed to a dict when it's JSON), so checks
20
+ can assert on the *effect* — e.g. a reorder that actually succeeded — not
21
+ merely that the tool was called.
22
+ """
23
+ tool: str
24
+ args: Dict = field(default_factory=dict)
25
+ result: object = None
26
+
27
+
28
+ @dataclass
29
+ class AgentRun:
30
+ """What an agent returned for one input: its reply plus the actions it took.
31
+
32
+ Adapters wrap a user's agent to produce this. ``actions`` is the part
33
+ text-only evals ignore and the part that catches "said right, did wrong".
34
+ """
35
+ output: str
36
+ actions: List[Action] = field(default_factory=list)
37
+
38
+ def actions_for(self, tool: str) -> List[Action]:
39
+ return [a for a in self.actions if a.tool == tool]
40
+
41
+
42
+ @dataclass
43
+ class Check:
44
+ """A single assertion about a run, tagged with the governance control it evidences."""
45
+ kind: str
46
+ describe: str
47
+ control: str
48
+ fn: Callable[[AgentRun], Tuple[bool, str]]
49
+
50
+
51
+ @dataclass
52
+ class CheckResult:
53
+ check: Check
54
+ passed: bool
55
+ detail: str
56
+ culprits: List[Action] = field(default_factory=list) # the exact trace steps at fault
57
+
58
+
59
+ @dataclass
60
+ class Scenario:
61
+ """An input plus the checks its run must satisfy — one test case."""
62
+ name: str
63
+ input: str
64
+ checks: List[Check] = field(default_factory=list)
65
+
66
+
67
+ @dataclass
68
+ class ScenarioResult:
69
+ scenario: Scenario
70
+ run: AgentRun
71
+ results: List[CheckResult]
72
+
73
+ @property
74
+ def passed(self) -> bool:
75
+ return all(r.passed for r in self.results)
76
+
77
+ def blame(self) -> Dict[int, List[str]]:
78
+ """Map trace-step index -> labels of the failed checks that flagged it.
79
+
80
+ This is the 'what went wrong' link: it points from a red check straight
81
+ to the exact action(s) in the trace that caused it.
82
+ """
83
+ by_id = {id(a): i for i, a in enumerate(self.run.actions)}
84
+ blame: Dict[int, List[str]] = {}
85
+ for r in self.results:
86
+ if r.passed:
87
+ continue
88
+ for a in r.culprits:
89
+ idx = by_id.get(id(a))
90
+ if idx is not None:
91
+ blame.setdefault(idx, []).append(r.check.describe)
92
+ return blame
93
+
94
+
95
+ @dataclass
96
+ class Report:
97
+ scenario_results: List[ScenarioResult]
98
+
99
+ @property
100
+ def total_checks(self) -> int:
101
+ return sum(len(s.results) for s in self.scenario_results)
102
+
103
+ @property
104
+ def failed_checks(self) -> int:
105
+ return sum(1 for s in self.scenario_results for r in s.results if not r.passed)
106
+
107
+ @property
108
+ def passed_scenarios(self) -> int:
109
+ return sum(1 for s in self.scenario_results if s.passed)
110
+
111
+ def control_rollup(self) -> Dict[str, Tuple[int, int]]:
112
+ """{control: (passed, total)} across every check, for the governance view."""
113
+ roll: Dict[str, List[int]] = {}
114
+ for s in self.scenario_results:
115
+ for r in s.results:
116
+ slot = roll.setdefault(r.check.control, [0, 0])
117
+ slot[1] += 1
118
+ if r.passed:
119
+ slot[0] += 1
120
+ return {k: (v[0], v[1]) for k, v in roll.items()}
121
+
122
+
123
+ def evaluate(agent_fn: Callable[[str], AgentRun], scenarios: List[Scenario]) -> Report:
124
+ """Run every scenario through the agent and score its checks.
125
+
126
+ ``agent_fn`` is the adapter: it takes the scenario input and returns an
127
+ ``AgentRun`` (reply + actions). Kept deliberately tiny so any framework —
128
+ pydantic-ai, LangChain, a raw API loop — can be wrapped in a few lines.
129
+ """
130
+ scenario_results: List[ScenarioResult] = []
131
+ for sc in scenarios:
132
+ run = agent_fn(sc.input)
133
+ results = []
134
+ for check in sc.checks:
135
+ culprits: List[Action] = []
136
+ try:
137
+ out = check.fn(run)
138
+ if len(out) == 3: # (ok, detail, culprit_actions)
139
+ ok, detail, culprits = out
140
+ else: # (ok, detail)
141
+ ok, detail = out
142
+ except Exception as exc: # a broken check must not abort the suite
143
+ ok, detail = False, f"check error: {exc}"
144
+ results.append(CheckResult(check, ok, detail, list(culprits or [])))
145
+ scenario_results.append(ScenarioResult(sc, run, results))
146
+ return Report(scenario_results)
agentlahon/judge.py ADDED
@@ -0,0 +1,139 @@
1
+ """LLM-as-judge — for the subjective checks code assertions can't cover.
2
+
3
+ Done the way the evals discipline demands (Hamel Husain / Shreya Shankar):
4
+ * **binary** verdicts, not Likert scores;
5
+ * an optional **domain reference** the judge grades against (reference-based);
6
+ * an **alignment** step that measures the judge's agreement with human labels
7
+ (TPR/TNR/accuracy/Cohen's κ) — because an unaligned judge is worse than no
8
+ judge. You don't trust it until it agrees with you.
9
+
10
+ The LLM is pluggable: pass any ``complete(prompt) -> str``. `openai_complete`
11
+ builds one from the OpenAI SDK (works with the OpenAI or a compatible proxy),
12
+ but the core stays dependency-free.
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import json
17
+ import os
18
+ import re
19
+ from typing import Callable, Dict, List, Tuple
20
+
21
+ from .core import AgentRun, Check
22
+ from .checks import CONTROLS
23
+
24
+ Complete = Callable[[str], str]
25
+
26
+ _PROMPT = '''You are a strict evaluator. Decide whether the ASSISTANT REPLY satisfies the CRITERION.
27
+ {reference}CRITERION: {question}
28
+
29
+ ASSISTANT REPLY:
30
+ """{output}"""
31
+
32
+ Respond with ONE JSON object and nothing else: {{"pass": true or false, "reason": "<one short sentence>"}}.'''
33
+
34
+
35
+ def _extract_verdict(raw: str) -> Tuple[bool, str]:
36
+ """Parse the judge's JSON verdict; degrade gracefully to keyword scan."""
37
+ try:
38
+ m = re.search(r"\{.*\}", raw, re.DOTALL)
39
+ obj = json.loads(m.group(0)) if m else json.loads(raw)
40
+ return bool(obj.get("pass")), str(obj.get("reason", ""))[:200]
41
+ except Exception:
42
+ low = raw.lower()
43
+ passed = ("true" in low or "pass" in low) and "false" not in low[:40]
44
+ return passed, raw.strip()[:160]
45
+
46
+
47
+ def llm_judge(question: str, complete: Complete, reference: str = "") -> Callable[[AgentRun], Tuple[bool, str]]:
48
+ """Return a judge fn (usable via ``expect.faithful`` / ``expect.judge``).
49
+
50
+ ``reference`` is the optional domain doc/policy/spec the reply is graded
51
+ against — left empty, the judge grades on the criterion alone.
52
+ """
53
+ ref_block = (f'REFERENCE (source of truth):\n"""{reference.strip()}"""\n\n'
54
+ if reference else "")
55
+
56
+ def judge(run: AgentRun) -> Tuple[bool, str]:
57
+ prompt = _PROMPT.format(reference=ref_block, question=question, output=run.output)
58
+ return _extract_verdict(complete(prompt))
59
+
60
+ return judge
61
+
62
+
63
+ def openai_complete(model: str = "gpt-4o-mini", base_url: str = None,
64
+ api_key: str = None, temperature: float = 0.0) -> Complete:
65
+ """Build a `complete()` from the OpenAI SDK. Optional dependency.
66
+
67
+ Reads the key from OPENAI_API_KEY / UDACITY_OPENAI_API_KEY if not passed.
68
+ Pass base_url to target a compatible proxy (e.g. the Vocareum endpoint).
69
+ """
70
+ from openai import OpenAI # optional; only needed for a live judge
71
+ client = OpenAI(base_url=base_url,
72
+ api_key=api_key or os.environ.get("OPENAI_API_KEY")
73
+ or os.environ.get("UDACITY_OPENAI_API_KEY"))
74
+
75
+ def complete(prompt: str) -> str:
76
+ r = client.chat.completions.create(
77
+ model=model, temperature=temperature, max_tokens=200,
78
+ messages=[{"role": "user", "content": prompt}])
79
+ return r.choices[0].message.content or ""
80
+
81
+ return complete
82
+
83
+
84
+ # --- Judge alignment (the Shreya rigor) --------------------------------------
85
+ def _cohens_kappa(tp: int, tn: int, fp: int, fn: int) -> float:
86
+ n = tp + tn + fp + fn
87
+ if n == 0:
88
+ return float("nan")
89
+ po = (tp + tn) / n
90
+ pe = ((tp + fp) * (tp + fn) + (fn + tn) * (fp + tn)) / (n * n)
91
+ return (po - pe) / (1 - pe) if pe != 1 else float("nan")
92
+
93
+
94
+ def align(judge: Callable[[AgentRun], Tuple[bool, str]],
95
+ labeled: List[Tuple[str, bool]]) -> Dict:
96
+ """Score a judge against human labels.
97
+
98
+ ``labeled`` is [(reply_text, human_pass), ...]. Returns accuracy, TPR
99
+ (catches real passes), TNR (catches real fails), Cohen's κ, and the
100
+ confusion counts. Treat a judge with low TNR as untrustworthy — it rubber-
101
+ stamps failures.
102
+ """
103
+ tp = tn = fp = fn = 0
104
+ disagreements = []
105
+ for text, human in labeled:
106
+ pred, reason = judge(AgentRun(output=text))
107
+ if human and pred:
108
+ tp += 1
109
+ elif human and not pred:
110
+ fn += 1; disagreements.append((text, human, pred, reason))
111
+ elif not human and pred:
112
+ fp += 1; disagreements.append((text, human, pred, reason))
113
+ else:
114
+ tn += 1
115
+ n = len(labeled)
116
+ return {
117
+ "n": n,
118
+ "accuracy": (tp + tn) / n if n else float("nan"),
119
+ "tpr": tp / (tp + fn) if (tp + fn) else float("nan"),
120
+ "tnr": tn / (tn + fp) if (tn + fp) else float("nan"),
121
+ "kappa": _cohens_kappa(tp, tn, fp, fn),
122
+ "tp": tp, "tn": tn, "fp": fp, "fn": fn,
123
+ "disagreements": disagreements,
124
+ }
125
+
126
+
127
+ def print_alignment(m: Dict) -> None:
128
+ def pct(x):
129
+ return "n/a" if x != x else f"{x*100:.0f}%" # x!=x catches NaN
130
+ print(f"Judge alignment on {m['n']} labeled examples:")
131
+ print(f" accuracy {pct(m['accuracy'])} TPR {pct(m['tpr'])} "
132
+ f"TNR {pct(m['tnr'])} κ {m['kappa']:.2f}")
133
+ print(f" confusion: tp={m['tp']} tn={m['tn']} fp={m['fp']} fn={m['fn']}")
134
+ verdict = ("TRUSTWORTHY" if m["kappa"] >= 0.6 and (m["tnr"] != m["tnr"] or m["tnr"] >= 0.7)
135
+ else "NEEDS WORK — iterate the judge prompt before relying on it")
136
+ print(f" -> {verdict}")
137
+ for text, human, pred, reason in m["disagreements"][:5]:
138
+ print(f" ✗ human={'pass' if human else 'fail'} judge={'pass' if pred else 'fail'}"
139
+ f" {text[:60]!r} ({reason[:50]})")
agentlahon/py.typed ADDED
File without changes
agentlahon/report.py ADDED
@@ -0,0 +1,144 @@
1
+ """Renderers for a Report — a terminal summary and a self-contained HTML file.
2
+
3
+ The HTML doubles as the shareable assurance artifact: a per-scenario pass/fail
4
+ view plus a governance rollup that maps every check to its NIST AI RMF control.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import html
9
+
10
+ from .core import Action, Report
11
+
12
+ _GREEN, _RED, _DIM, _BOLD, _RESET = "\033[32m", "\033[31m", "\033[2m", "\033[1m", "\033[0m"
13
+ _YEL = "\033[33m"
14
+
15
+
16
+ def _fmt_args(d: dict) -> str:
17
+ items = [f"{k}={v}" for k, v in d.items()][:4]
18
+ return ", ".join(items)
19
+
20
+
21
+ def _fmt_result(res) -> str:
22
+ """Surface the telling bits of a tool's return (ordered/recorded/etc.)."""
23
+ if isinstance(res, dict):
24
+ for k in ("ordered", "recorded", "sufficient", "carried", "reason"):
25
+ if k in res:
26
+ return f"{k}={res[k]}"
27
+ return ""
28
+
29
+
30
+ def _fmt_action(a: Action) -> str:
31
+ tail = _fmt_result(a.result)
32
+ return f"{a.tool}({_fmt_args(a.args)})" + (f" → {tail}" if tail else "")
33
+
34
+
35
+ def print_terminal(report: Report, trace: bool = True) -> None:
36
+ for s in report.scenario_results:
37
+ head = f"{_GREEN}PASS{_RESET}" if s.passed else f"{_RED}FAIL{_RESET}"
38
+ print(f"\n{head} {_BOLD}{s.scenario.name}{_RESET}")
39
+ print(f" {_DIM}input:{_RESET} {s.scenario.input[:80]}")
40
+ for r in s.results:
41
+ mark = f"{_GREEN}✓{_RESET}" if r.passed else f"{_RED}✗{_RESET}"
42
+ print(f" {mark} {r.check.describe:<34} {_DIM}{r.detail}{_RESET}")
43
+
44
+ # Trace: show what the agent DID, pinpointing the step(s) at fault.
45
+ if trace and s.run.actions and not s.passed:
46
+ blame = s.blame()
47
+ print(f" {_DIM}trace (what the agent did):{_RESET}")
48
+ for i, a in enumerate(s.run.actions):
49
+ if i in blame:
50
+ why = "; ".join(blame[i])
51
+ print(f" {_RED}✗ {i+1}. {_fmt_action(a)}{_RESET}"
52
+ f" {_YEL}← {why}{_RESET}")
53
+ else:
54
+ print(f" {_DIM}· {i+1}. {_fmt_action(a)}{_RESET}")
55
+
56
+ sc_pass, sc_tot = report.passed_scenarios, len(report.scenario_results)
57
+ ck_fail, ck_tot = report.failed_checks, report.total_checks
58
+ print(f"\n{_BOLD}Scenarios:{_RESET} {sc_pass}/{sc_tot} passed "
59
+ f"{_BOLD}Checks:{_RESET} {ck_tot - ck_fail}/{ck_tot} passed")
60
+ print(f"{_BOLD}Governance rollup (NIST AI RMF):{_RESET}")
61
+ for control, (p, t) in sorted(report.control_rollup().items()):
62
+ bar = f"{_GREEN}OK{_RESET}" if p == t else f"{_RED}{t - p} gap(s){_RESET}"
63
+ print(f" {control:<44} {p}/{t} {bar}")
64
+
65
+
66
+ def write_html(report: Report, path: str = "assurance_report.html") -> str:
67
+ rows = []
68
+ for s in report.scenario_results:
69
+ badge = "pass" if s.passed else "fail"
70
+ checks = "".join(
71
+ f'<li class="{ "ok" if r.passed else "no" }"><span>{"✓" if r.passed else "✗"}</span>'
72
+ f'<b>{html.escape(r.check.describe)}</b>'
73
+ f'<code>{html.escape(r.check.control)}</code>'
74
+ f'<em>{html.escape(r.detail)}</em></li>'
75
+ for r in s.results)
76
+ trace_html = ""
77
+ if s.run.actions:
78
+ blame = s.blame()
79
+ steps = "".join(
80
+ f'<li class="{ "no" if i in blame else "" }">'
81
+ f'<span>{i+1}</span><code>{html.escape(_fmt_action(a))}</code>'
82
+ + (f'<em>← {html.escape("; ".join(blame[i]))}</em>' if i in blame else '')
83
+ + '</li>'
84
+ for i, a in enumerate(s.run.actions))
85
+ trace_html = f'<details class="trace"{" open" if not s.passed else ""}>' \
86
+ f'<summary>trace · {len(s.run.actions)} actions</summary>' \
87
+ f'<ol>{steps}</ol></details>'
88
+ rows.append(
89
+ f'<section class="sc {badge}"><h3><span class="tag">{badge.upper()}</span>'
90
+ f'{html.escape(s.scenario.name)}</h3>'
91
+ f'<p class="in">{html.escape(s.scenario.input)}</p><ul>{checks}</ul>{trace_html}</section>')
92
+
93
+ roll = "".join(
94
+ f'<tr class="{ "ok" if p==t else "no" }"><td>{html.escape(c)}</td>'
95
+ f'<td>{p}/{t}</td><td>{"Conformant" if p==t else f"{t-p} gap(s)"}</td></tr>'
96
+ for c, (p, t) in sorted(report.control_rollup().items()))
97
+
98
+ sc_pass, sc_tot = report.passed_scenarios, len(report.scenario_results)
99
+ ck_pass, ck_tot = report.total_checks - report.failed_checks, report.total_checks
100
+ doc = f"""<!doctype html><meta charset="utf-8"><title>AI Assurance Report</title>
101
+ <style>
102
+ :root{{--bg:#0b0d10;--card:#15181d;--line:#262b33;--dim:#8b95a5;--ok:#2fbf71;--no:#ef4d5a;--fg:#e8ecf1}}
103
+ *{{box-sizing:border-box}}body{{font:15px/1.5 -apple-system,Segoe UI,Roboto,sans-serif;background:var(--bg);color:var(--fg);margin:0;padding:32px;max-width:900px;margin:auto}}
104
+ h1{{font-size:22px;margin:0 0 4px}}.sub{{color:var(--dim);margin:0 0 24px}}
105
+ .kpis{{display:flex;gap:12px;margin-bottom:28px;flex-wrap:wrap}}
106
+ .kpi{{background:var(--card);border:1px solid var(--line);border-radius:12px;padding:14px 18px;flex:1;min-width:150px}}
107
+ .kpi b{{display:block;font-size:26px}}.kpi span{{color:var(--dim);font-size:13px}}
108
+ table{{width:100%;border-collapse:collapse;background:var(--card);border:1px solid var(--line);border-radius:12px;overflow:hidden;margin-bottom:28px}}
109
+ td,th{{text-align:left;padding:10px 14px;border-bottom:1px solid var(--line);font-size:14px}}th{{color:var(--dim);font-weight:600}}
110
+ tr.no td:last-child{{color:var(--no)}}tr.ok td:last-child{{color:var(--ok)}}
111
+ .sc{{background:var(--card);border:1px solid var(--line);border-left:4px solid var(--ok);border-radius:12px;padding:14px 18px;margin-bottom:14px}}
112
+ .sc.fail{{border-left-color:var(--no)}}
113
+ .sc h3{{margin:0 0 6px;font-size:16px;display:flex;align-items:center;gap:10px}}
114
+ .tag{{font-size:11px;padding:2px 8px;border-radius:20px;background:rgba(47,191,113,.15);color:var(--ok)}}
115
+ .fail .tag{{background:rgba(239,77,90,.15);color:var(--no)}}
116
+ .in{{color:var(--dim);font-size:13px;margin:0 0 10px}}
117
+ ul{{list-style:none;margin:0;padding:0}}li{{display:grid;grid-template-columns:20px 1fr auto;gap:8px;align-items:center;padding:6px 0;border-top:1px solid var(--line);font-size:13px}}
118
+ li span{{font-weight:700}}li.ok span{{color:var(--ok)}}li.no span{{color:var(--no)}}
119
+ li code{{color:var(--dim);font-size:11px;grid-column:2;justify-self:start}}li em{{grid-column:3;color:var(--dim);font-style:normal;font-size:12px}}
120
+ li b{{font-weight:500}}
121
+ .trace{{margin-top:10px;border-top:1px solid var(--line);padding-top:8px}}
122
+ .trace summary{{color:var(--dim);font-size:12px;cursor:pointer;list-style:revert}}
123
+ .trace ol{{margin:8px 0 0;padding-left:0;list-style:none;counter-reset:step}}
124
+ .trace ol li{{display:grid;grid-template-columns:20px auto 1fr;gap:8px;font-family:ui-monospace,SFMono-Regular,Menlo,monospace}}
125
+ .trace ol li span{{color:var(--dim);font-weight:600}}
126
+ .trace ol li.no{{background:rgba(239,77,90,.07);border-radius:6px}}
127
+ .trace ol li.no span,.trace ol li.no em{{color:var(--no)}}
128
+ .trace ol li code{{color:var(--fg);grid-column:2}}.trace ol li em{{grid-column:3;font-style:normal}}
129
+ </style>
130
+ <h1>AI Assurance Report</h1>
131
+ <p class="sub">Behavioral evaluation of an AI agent · say-level + do-level checks · mapped to NIST AI RMF</p>
132
+ <div class="kpis">
133
+ <div class="kpi"><b>{sc_pass}/{sc_tot}</b><span>scenarios passed</span></div>
134
+ <div class="kpi"><b>{ck_pass}/{ck_tot}</b><span>checks passed</span></div>
135
+ <div class="kpi"><b>{report.failed_checks}</b><span>open findings</span></div>
136
+ </div>
137
+ <h2 style="font-size:16px">Governance conformance</h2>
138
+ <table><tr><th>Control (NIST AI RMF)</th><th>Passed</th><th>Status</th></tr>{roll}</table>
139
+ <h2 style="font-size:16px">Scenario detail</h2>
140
+ {''.join(rows)}
141
+ """
142
+ with open(path, "w") as f:
143
+ f.write(doc)
144
+ return path
agentlahon/store.py ADDED
@@ -0,0 +1,95 @@
1
+ """Trace store — persist agent runs so you can *look at your data*.
2
+
3
+ The evals discipline (Hamel Husain / Shreya Shankar) starts with error
4
+ analysis: log real traces, read them, tag the failures, and let those tagged
5
+ failures drive your assertions. This module is that on-ramp — an append-only
6
+ JSONL log of runs plus their human annotations.
7
+
8
+ store = TraceStore("traces.jsonl")
9
+ tid = store.record("500 flyers", run) # log a real agent run
10
+ store.annotate(tid, "fail", label="sold-noncarried", note="flyers aren't stocked")
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import os
16
+ from dataclasses import asdict, dataclass, field
17
+ from typing import Dict, List, Optional
18
+
19
+ from .core import Action, AgentRun
20
+
21
+
22
+ @dataclass
23
+ class Annotation:
24
+ verdict: Optional[str] = None # "pass" | "fail" | None (untagged)
25
+ label: str = "" # short failure-mode tag (open coding)
26
+ note: str = ""
27
+
28
+
29
+ @dataclass
30
+ class TraceRecord:
31
+ id: str
32
+ input: str
33
+ output: str
34
+ actions: List[Action] = field(default_factory=list)
35
+ annotation: Annotation = field(default_factory=Annotation)
36
+
37
+ @property
38
+ def tagged(self) -> bool:
39
+ return self.annotation.verdict is not None
40
+
41
+
42
+ def _record_to_dict(r: TraceRecord) -> Dict:
43
+ return {"id": r.id, "input": r.input, "output": r.output,
44
+ "actions": [asdict(a) for a in r.actions],
45
+ "annotation": asdict(r.annotation)}
46
+
47
+
48
+ def _record_from_dict(d: Dict) -> TraceRecord:
49
+ return TraceRecord(
50
+ id=d["id"], input=d.get("input", ""), output=d.get("output", ""),
51
+ actions=[Action(**a) for a in d.get("actions", [])],
52
+ annotation=Annotation(**d.get("annotation", {})))
53
+
54
+
55
+ class TraceStore:
56
+ """Append-only JSONL log of runs + annotations. Rewritten on annotate."""
57
+
58
+ def __init__(self, path: str):
59
+ self.path = path
60
+ self.records: List[TraceRecord] = []
61
+ if os.path.exists(path):
62
+ with open(path) as f:
63
+ self.records = [_record_from_dict(json.loads(line))
64
+ for line in f if line.strip()]
65
+
66
+ def _next_id(self) -> str:
67
+ return f"t{len(self.records) + 1:04d}"
68
+
69
+ def record(self, input: str, run: AgentRun) -> str:
70
+ rec = TraceRecord(self._next_id(), input, run.output, list(run.actions))
71
+ self.records.append(rec)
72
+ with open(self.path, "a") as f: # append the new run
73
+ f.write(json.dumps(_record_to_dict(rec)) + "\n")
74
+ return rec.id
75
+
76
+ def get(self, tid: str) -> Optional[TraceRecord]:
77
+ return next((r for r in self.records if r.id == tid), None)
78
+
79
+ def annotate(self, tid: str, verdict: str, label: str = "", note: str = "") -> None:
80
+ rec = self.get(tid)
81
+ if rec is None:
82
+ raise KeyError(tid)
83
+ rec.annotation = Annotation(verdict=verdict, label=label, note=note)
84
+ self._rewrite()
85
+
86
+ def untagged(self) -> List[TraceRecord]:
87
+ return [r for r in self.records if not r.tagged]
88
+
89
+ def failures(self) -> List[TraceRecord]:
90
+ return [r for r in self.records if r.annotation.verdict == "fail"]
91
+
92
+ def _rewrite(self) -> None:
93
+ with open(self.path, "w") as f:
94
+ for r in self.records:
95
+ f.write(json.dumps(_record_to_dict(r)) + "\n")
@@ -0,0 +1,187 @@
1
+ Metadata-Version: 2.4
2
+ Name: agentlahon
3
+ Version: 0.0.1
4
+ Summary: Unit tests for AI agents — catch when your agent does the wrong thing, not just when it says the wrong thing.
5
+ Author-email: Anurag Lahon <anuraglahondp@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/anuraglahon16/agentcheck
8
+ Project-URL: Repository, https://github.com/anuraglahon16/agentcheck
9
+ Project-URL: Issues, https://github.com/anuraglahon16/agentcheck/issues
10
+ Keywords: ai,agents,evaluation,evals,llm,testing,governance
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Topic :: Software Development :: Testing
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.9
23
+ Description-Content-Type: text/markdown
24
+ License-File: LICENSE
25
+ Provides-Extra: dev
26
+ Requires-Dist: pytest>=7; extra == "dev"
27
+ Provides-Extra: judge
28
+ Requires-Dist: openai>=1; extra == "judge"
29
+ Provides-Extra: webapp
30
+ Requires-Dist: flask>=3; extra == "webapp"
31
+ Dynamic: license-file
32
+
33
+ # agentlahon
34
+
35
+ ### Unit tests for AI agents — catch when your agent *does* the wrong thing, not just when it *says* the wrong thing.
36
+
37
+ ![tests](https://img.shields.io/badge/tests-passing-2fbf71) ![python](https://img.shields.io/badge/python-3.9%2B-blue) ![license](https://img.shields.io/badge/license-MIT-black) ![deps](https://img.shields.io/badge/core%20deps-zero-black)
38
+
39
+ > **Status:** v0.0.1, experimental — APIs may change. Core is zero-dependency. Feedback and first users very welcome.
40
+
41
+ Text-only evals grade what an agent *says*. But agents take *actions* — call tools, write to databases, move money. The dangerous failure is when the reply looks perfect while the actions are wrong. **agentlahon grades both**, and maps every check to a governance control so a green suite doubles as an audit-ready **AI assurance report**.
42
+
43
+ ```
44
+ FAIL decline-only-noncarried
45
+ ✓ reply mentions 'unable' found ← SAID the right thing
46
+ ✓ never records a sale no such action ✓
47
+ ✗ never restocks (money moves) took 1×: reorder Invitation-cards ×9000 ← DID the wrong thing
48
+ ```
49
+
50
+ > That failure is invisible to every text-based eval. Only checking the agent's **actions** catches it. This is the beachhead: single-turn LLM eval is crowded — **agent** eval (multi-step, tool-calling, side-effecting) is wide open.
51
+
52
+ ```python
53
+ Scenario("decline-only", "5000 flyers, 10000 tickets", checks=[
54
+ expect.output_contains("unable"), # say-level
55
+ expect.no_action("reorder_item", # do-level — the part text evals miss
56
+ where=lambda a: a.result.get("ordered")), # assert on the *effect*, not just the call
57
+ ])
58
+ ```
59
+
60
+ ## Install
61
+
62
+ ```bash
63
+ pip install agentlahon # PyPI distribution name
64
+ pip install -e . # or from source
65
+ ```
66
+
67
+ Core is dependency-free. Python ≥ 3.9.
68
+
69
+ ## Quickstart
70
+
71
+ ```python
72
+ from agentlahon import Scenario, expect, evaluate, print_terminal, write_html
73
+
74
+ def agent(text): # wrap YOUR agent -> AgentRun(output, actions)
75
+ ...
76
+
77
+ scenarios = [
78
+ Scenario("decline-noncarried", "5000 flyers, 10000 tickets", checks=[
79
+ expect.output_contains("unable"),
80
+ expect.no_action("reorder_item"), # the do-level check text evals miss
81
+ expect.no_pii(),
82
+ ]),
83
+ ]
84
+ report = evaluate(agent, scenarios)
85
+ print_terminal(report)
86
+ write_html(report) # shareable assurance_report.html
87
+ ```
88
+
89
+ ## The evals flywheel — look at data → tag → generate assertions
90
+
91
+ The hard part of evals isn't running assertions, it's *knowing what to assert*. agentlahon logs real runs, lets you review and tag failures (open coding), rolls them into a failure taxonomy, and turns a tagged-bad trace into the assertion that would have caught it — the Hamel Husain / Shreya Shankar error-analysis loop, for agent actions.
92
+
93
+ ```bash
94
+ agentlahon run suite.py --log traces.jsonl # 1. log real runs
95
+ agentlahon review traces.jsonl # 2. page through, tag failures
96
+ agentlahon analyze traces.jsonl # 3. failure taxonomy (what dominates)
97
+ agentlahon suggest traces.jsonl t0003 # 4. trace -> ready-to-paste assertion
98
+ ```
99
+
100
+ ```
101
+ [t0003] input: 5000 flyers, 2000 posters, 10000 tickets
102
+ trace: 1. reorder_item(Invitation cards, 9000) → ordered=True
103
+ suggested assertions:
104
+ expect.no_action("reorder_item", where=lambda a: (a.result or {}).get("ordered"))
105
+ # reorder_item took effect — guard against it when it should not fire
106
+ ```
107
+
108
+ ## LLM-as-judge — for subjective checks, aligned before you trust it
109
+
110
+ Code assertions can't judge "is this reply faithful / on-policy / correct for our domain?" — that needs an LLM judge. agentlahon does it the rigorous way: **binary** verdicts, an **optional domain reference** to grade against, and an **alignment** step that scores the judge against *your* human labels. An unaligned judge is worse than none.
111
+
112
+ ```python
113
+ from agentlahon import expect, openai_complete, llm_judge, align, print_alignment
114
+
115
+ complete = openai_complete(model="gpt-4o-mini") # pluggable; pip install "agentlahon[judge]"
116
+
117
+ check = expect.judge("Does the reply stay within our refund policy?",
118
+ complete=complete,
119
+ reference=open("refund_policy.md").read()) # optional domain doc
120
+
121
+ # Don't trust the judge until it agrees with you:
122
+ labeled = [("we'll refund within 30 days", True), ("sure, full refund anytime", False), ...]
123
+ print_alignment(align(llm_judge("within policy?", complete), labeled))
124
+ # accuracy 92% TPR 95% TNR 88% κ 0.83 -> TRUSTWORTHY
125
+ ```
126
+
127
+ A judge that rubber-stamps everything scores **TNR 0% → NEEDS WORK** — caught before it hides real failures.
128
+
129
+ ## Traces — *what* went wrong, not just *that* it did
130
+
131
+ When a check fails, agentlahon prints the agent's action trace and points at the exact step:
132
+
133
+ ```
134
+ ✗ never restocks (money moves) took 1×: [Invitation cards ×9000]
135
+ trace (what the agent did):
136
+ · 1. tool_reorder_item(Flyers, 5050) → ordered=False (refused, harmless)
137
+ · 2. tool_reorder_item(Poster paper, 2050) → ordered=False (refused, harmless)
138
+ ✗ 3. tool_reorder_item(Invitation cards, 9000)→ ordered=True ← never restocks (money moves)
139
+ ```
140
+
141
+ The failing check is linked to the offending step (`ScenarioResult.blame()`), so you go from red to root cause instantly. Same trace renders in the HTML report.
142
+
143
+ ## CLI (drop into CI)
144
+
145
+ A suite file defines `scenarios` and `agent`; the CLI exits non-zero on findings:
146
+
147
+ ```bash
148
+ agentlahon run examples/suite.py --html report.html
149
+ ```
150
+
151
+ ## Run the examples (no API key)
152
+
153
+ ```bash
154
+ python examples/run_demo.py # synthetic agent with a planted bug
155
+ python examples/run_beavers.py # points at a REAL pydantic-ai agent, catches a real regression
156
+ pytest # 6 core tests
157
+ ```
158
+
159
+ ## Layout
160
+
161
+ ```
162
+ src/agentlahon/ core · checks · adapters · report · cli
163
+ examples/ run_demo · run_beavers · suite
164
+ tests/ test_core
165
+ ```
166
+
167
+ ## Checks
168
+
169
+ | Family | Checks | Control (NIST AI RMF) |
170
+ |---|---|---|
171
+ | say-level | `output_contains`, `output_absent` | MEASURE-2.3 Task performance |
172
+ | **do-level** | `no_action`, `action_taken`, `max_actions`, `actions_only_on` | MANAGE-2.1 Action safety |
173
+ | privacy | `no_pii` | MEASURE-2.10 Privacy |
174
+ | transparency | `no_internal_leak` | MEASURE-2.9 Transparency |
175
+ | faithfulness | `faithful(judge)` — plug in LLM-as-judge | MEASURE-2.5 Validity |
176
+
177
+ ## Roadmap (v0 → product)
178
+
179
+ - [x] Function-capture adapter (auto-records real tool calls **and effects**)
180
+ - [x] CI integration (`agentlahon run` exits non-zero on findings)
181
+ - [ ] Native adapters for OpenAI/Anthropic tool-calls & LangGraph traces
182
+ - [ ] LLM-as-judge faithfulness + bias checks
183
+ - [ ] Regression mode: diff a run against a saved baseline on model/prompt change
184
+ - [ ] Hosted dashboard + shareable report links (the paid layer)
185
+
186
+ Status: **v0.0.1** — installable package, CLI, effect-level checks, terminal + HTML
187
+ report, tests, and a working run against a real pydantic-ai agent.
@@ -0,0 +1,16 @@
1
+ agentlahon/__init__.py,sha256=6U_b00npJ-JDRG2jsV0kcY3ftZEpCkyIsX2ZGk2ojHQ,1320
2
+ agentlahon/adapters.py,sha256=cEnJirnnyBel9GOzb5KQlWLA3Mf9UBjENBswsLUSFsE,2565
3
+ agentlahon/analysis.py,sha256=NZaR6gd-4tU5iD1t95JcplCEJpG9bXcR0JXJ1BLi8UE,2870
4
+ agentlahon/checks.py,sha256=UWSAgTTFEa1UAs4QIwCpA-izaLE4Bq-nWrvf5JQL-8g,6171
5
+ agentlahon/cli.py,sha256=OJKFjIGJEAc_fxRiu3qHhVJH0qCXaoGwfasNoLTg3MI,5651
6
+ agentlahon/core.py,sha256=Imnzo6jX8ogknLOYC7UbgI-JACqbWYsuVcgaqj4kw2o,4933
7
+ agentlahon/judge.py,sha256=KCpya4cXZDGQUkltjBe6mIhlIVRPl733WLL2vuv_4_g,5655
8
+ agentlahon/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
9
+ agentlahon/report.py,sha256=WdDLSdCV8a8VYOL3nvzs7rfqTTgbIxLLvr3rOlijZHw,7804
10
+ agentlahon/store.py,sha256=DWSqq4Hp8mgSdVig31batvd0c6NPElwsszG_0pEFDJo,3305
11
+ agentlahon-0.0.1.dist-info/licenses/LICENSE,sha256=3MDlg5-4nqpRTZIameTXgo5kxGeWsXdup3HWiNj-qFQ,1080
12
+ agentlahon-0.0.1.dist-info/METADATA,sha256=5Bo8OR0hDHutQTqVQpod0xcaN5PsRH4vAigz-WGvBkk,8416
13
+ agentlahon-0.0.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
14
+ agentlahon-0.0.1.dist-info/entry_points.txt,sha256=H4H2Vnanb_vzEgd6nv8t2ikJycB7h7noI_cypcYnuVM,51
15
+ agentlahon-0.0.1.dist-info/top_level.txt,sha256=tlm_aI7A2oTai1ijcImYweFx9NLEMdRMDkXSpHZcrh8,11
16
+ agentlahon-0.0.1.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ agentlahon = agentlahon.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 agentcheck contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ agentlahon