agentlahon 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentlahon/__init__.py +31 -0
- agentlahon/adapters.py +72 -0
- agentlahon/analysis.py +77 -0
- agentlahon/checks.py +127 -0
- agentlahon/cli.py +160 -0
- agentlahon/core.py +146 -0
- agentlahon/judge.py +139 -0
- agentlahon/py.typed +0 -0
- agentlahon/report.py +144 -0
- agentlahon/store.py +95 -0
- agentlahon-0.0.1.dist-info/METADATA +187 -0
- agentlahon-0.0.1.dist-info/RECORD +16 -0
- agentlahon-0.0.1.dist-info/WHEEL +5 -0
- agentlahon-0.0.1.dist-info/entry_points.txt +2 -0
- agentlahon-0.0.1.dist-info/licenses/LICENSE +21 -0
- agentlahon-0.0.1.dist-info/top_level.txt +1 -0
agentlahon/__init__.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""agentlahon — unit tests for AI agents.
|
|
2
|
+
|
|
3
|
+
Catch when your agent does the wrong thing, not just when it says the wrong
|
|
4
|
+
thing. Score say-level (reply) and do-level (actions/side-effects) behavior,
|
|
5
|
+
mapped to governance controls.
|
|
6
|
+
|
|
7
|
+
from agentlahon import Scenario, expect, evaluate, print_terminal
|
|
8
|
+
|
|
9
|
+
scenarios = [
|
|
10
|
+
Scenario("decline-noncarried", "5000 flyers, 10000 tickets", checks=[
|
|
11
|
+
expect.output_contains("unable"),
|
|
12
|
+
expect.no_action("reorder_item"), # <- the do-level check text evals miss
|
|
13
|
+
expect.no_pii(),
|
|
14
|
+
]),
|
|
15
|
+
]
|
|
16
|
+
report = evaluate(my_agent_adapter, scenarios)
|
|
17
|
+
print_terminal(report)
|
|
18
|
+
"""
|
|
19
|
+
from .core import Action, AgentRun, Scenario, Check, Report, evaluate
|
|
20
|
+
from .checks import expect, CONTROLS
|
|
21
|
+
from .report import print_terminal, write_html
|
|
22
|
+
from .adapters import capture
|
|
23
|
+
from .store import TraceStore, TraceRecord
|
|
24
|
+
from .analysis import taxonomy, suggest_checks
|
|
25
|
+
from .judge import llm_judge, openai_complete, align, print_alignment
|
|
26
|
+
|
|
27
|
+
__all__ = ["Action", "AgentRun", "Scenario", "Check", "Report", "evaluate",
|
|
28
|
+
"expect", "CONTROLS", "print_terminal", "write_html", "capture",
|
|
29
|
+
"TraceStore", "TraceRecord", "taxonomy", "suggest_checks",
|
|
30
|
+
"llm_judge", "openai_complete", "align", "print_alignment"]
|
|
31
|
+
__version__ = "0.0.1"
|
agentlahon/adapters.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Adapters that auto-capture an agent's actions so you don't wire them by hand.
|
|
2
|
+
|
|
3
|
+
`capture()` wraps a set of functions on a module (typically the tool functions
|
|
4
|
+
an agent calls) and records every invocation as an `Action`. Point it at your
|
|
5
|
+
real code, run the agent, and you get the `AgentRun.actions` list for free —
|
|
6
|
+
this is how agentlahon sees *what the agent did*, not just what it said.
|
|
7
|
+
|
|
8
|
+
Works with any framework whose tools bottom out in Python callables:
|
|
9
|
+
pydantic-ai, LangChain, or a hand-rolled loop. For pydantic-ai specifically,
|
|
10
|
+
the tool wrappers call module-level `tool_*` functions, so capturing those
|
|
11
|
+
records the real side-effects the model triggered.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
from contextlib import contextmanager
|
|
17
|
+
from typing import Dict, Iterable, List
|
|
18
|
+
|
|
19
|
+
from .core import Action
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _parse_result(value):
|
|
23
|
+
"""Parse a tool's return into a dict when possible, so checks can inspect it."""
|
|
24
|
+
if isinstance(value, dict):
|
|
25
|
+
return value
|
|
26
|
+
if isinstance(value, str):
|
|
27
|
+
try:
|
|
28
|
+
return json.loads(value)
|
|
29
|
+
except (ValueError, TypeError):
|
|
30
|
+
return value
|
|
31
|
+
return value
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _summarize(tool: str, args: tuple, kwargs: dict) -> Dict:
|
|
35
|
+
"""Best-effort, readable arg snapshot — keeps the report legible."""
|
|
36
|
+
snap = dict(kwargs)
|
|
37
|
+
for i, a in enumerate(args):
|
|
38
|
+
# skip long/opaque positional blobs (JSON strings, dates); keep short ones
|
|
39
|
+
if isinstance(a, (int, float)):
|
|
40
|
+
snap[f"arg{i}"] = a
|
|
41
|
+
elif isinstance(a, str) and len(a) <= 60:
|
|
42
|
+
snap[f"arg{i}"] = a
|
|
43
|
+
return snap
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@contextmanager
|
|
47
|
+
def capture(module, names: Iterable[str]):
|
|
48
|
+
"""Temporarily wrap ``module.<name>`` funcs to log calls; restore on exit.
|
|
49
|
+
|
|
50
|
+
Yields a live list of `Action`s. Callers that reference the wrapped names as
|
|
51
|
+
module globals (the usual case) will hit the wrapper, so real tool calls are
|
|
52
|
+
recorded without touching the agent's code.
|
|
53
|
+
"""
|
|
54
|
+
log: List[Action] = []
|
|
55
|
+
originals = {}
|
|
56
|
+
for name in names:
|
|
57
|
+
originals[name] = getattr(module, name)
|
|
58
|
+
|
|
59
|
+
def make(n, orig):
|
|
60
|
+
def wrapper(*args, **kwargs):
|
|
61
|
+
out = orig(*args, **kwargs)
|
|
62
|
+
log.append(Action(tool=n, args=_summarize(n, args, kwargs),
|
|
63
|
+
result=_parse_result(out)))
|
|
64
|
+
return out
|
|
65
|
+
return wrapper
|
|
66
|
+
|
|
67
|
+
setattr(module, name, make(name, originals[name]))
|
|
68
|
+
try:
|
|
69
|
+
yield log
|
|
70
|
+
finally:
|
|
71
|
+
for name, orig in originals.items():
|
|
72
|
+
setattr(module, name, orig)
|
agentlahon/analysis.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Error analysis + trace→assertion suggestions.
|
|
2
|
+
|
|
3
|
+
Two jobs from the evals playbook:
|
|
4
|
+
* `taxonomy()` — cluster tagged failures by their open-coded label, so you
|
|
5
|
+
see which failure modes dominate (the "axial coding" rollup).
|
|
6
|
+
* `suggest_checks()` — turn a tagged-bad trace into concrete `expect.*`
|
|
7
|
+
assertions that would have caught it. This closes the loop: look at data →
|
|
8
|
+
label failure → generate the eval.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Dict, List, Tuple
|
|
13
|
+
|
|
14
|
+
from .core import Action
|
|
15
|
+
from .store import TraceRecord, TraceStore
|
|
16
|
+
|
|
17
|
+
_EFFECT_KEYS = ("ordered", "recorded", "sold", "success", "created", "sent")
|
|
18
|
+
_ITEM_ARGS = ("item", "name", "product", "arg0")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def taxonomy(store: TraceStore) -> Dict[str, int]:
|
|
22
|
+
"""{failure-label: count} over tagged failures — most common first."""
|
|
23
|
+
counts: Dict[str, int] = {}
|
|
24
|
+
for r in store.failures():
|
|
25
|
+
label = r.annotation.label or "(unlabeled)"
|
|
26
|
+
counts[label] = counts.get(label, 0) + 1
|
|
27
|
+
return dict(sorted(counts.items(), key=lambda kv: -kv[1]))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _effect_key(a: Action) -> str:
|
|
31
|
+
if isinstance(a.result, dict):
|
|
32
|
+
for k in _EFFECT_KEYS:
|
|
33
|
+
if a.result.get(k):
|
|
34
|
+
return k
|
|
35
|
+
return ""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def suggest_checks(rec: TraceRecord) -> List[Tuple[str, str]]:
|
|
39
|
+
"""Propose (assertion_code, rationale) for a bad trace.
|
|
40
|
+
|
|
41
|
+
Heuristics over what the agent actually DID, so the suggestions are grounded
|
|
42
|
+
in this trace — not a generic checklist.
|
|
43
|
+
"""
|
|
44
|
+
out: List[Tuple[str, str]] = []
|
|
45
|
+
seen = set()
|
|
46
|
+
|
|
47
|
+
for a in rec.actions:
|
|
48
|
+
key = _effect_key(a)
|
|
49
|
+
if key and a.tool not in seen:
|
|
50
|
+
seen.add(a.tool)
|
|
51
|
+
out.append((
|
|
52
|
+
f'expect.no_action("{a.tool}", '
|
|
53
|
+
f'where=lambda a: (a.result or {{}}).get("{key}"))',
|
|
54
|
+
f'{a.tool} took effect ({key}=True) — guard against it when it should not fire',
|
|
55
|
+
))
|
|
56
|
+
|
|
57
|
+
# scope guard if actions carry an item-like arg
|
|
58
|
+
for a in rec.actions:
|
|
59
|
+
arg = next((k for k in _ITEM_ARGS if k in a.args), None)
|
|
60
|
+
if arg and (a.tool, "scope") not in seen:
|
|
61
|
+
seen.add((a.tool, "scope"))
|
|
62
|
+
out.append((
|
|
63
|
+
f'expect.actions_only_on("{a.tool}", "{arg}", ALLOWED)',
|
|
64
|
+
f'{a.tool} acted on {a.args.get(arg)!r} — restrict it to an allow-set',
|
|
65
|
+
))
|
|
66
|
+
break
|
|
67
|
+
|
|
68
|
+
# privacy guard if the reply looks like it leaked something
|
|
69
|
+
low = rec.output.lower()
|
|
70
|
+
if "@" in rec.output or any(t in low for t in ("margin", "cost", "internal")):
|
|
71
|
+
out.append(('expect.no_pii() # and expect.no_internal_leak()',
|
|
72
|
+
"reply may leak PII or internal info"))
|
|
73
|
+
|
|
74
|
+
if not out:
|
|
75
|
+
out.append(('expect.output_absent("<bad phrase>")',
|
|
76
|
+
"no action effects found — assert on the reply text instead"))
|
|
77
|
+
return out
|
agentlahon/checks.py
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Check factory — the assertions you attach to a Scenario.
|
|
2
|
+
|
|
3
|
+
Each `expect.*` returns a Check tagged with the NIST AI RMF control it
|
|
4
|
+
evidences, so a green suite *is* your assurance report. Checks split into
|
|
5
|
+
two families: **say-level** (about the reply text) and **do-level** (about the
|
|
6
|
+
actions taken) — the do-level ones are what text-only evals can't see.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import re
|
|
11
|
+
from typing import Iterable
|
|
12
|
+
|
|
13
|
+
from .core import AgentRun, Check
|
|
14
|
+
|
|
15
|
+
# --- Governance control mapping (NIST AI RMF subcategories) -------------------
|
|
16
|
+
CONTROLS = {
|
|
17
|
+
"TASK": "MEASURE-2.3 · Task performance",
|
|
18
|
+
"ACTION": "MANAGE-2.1 · Action safety & side-effects",
|
|
19
|
+
"PRIVACY": "MEASURE-2.10 · Privacy / data leakage",
|
|
20
|
+
"TRANSPARENCY": "MEASURE-2.9 · Transparency (no internal info)",
|
|
21
|
+
"ROBUSTNESS": "MEASURE-2.7 · Robustness to adversarial input",
|
|
22
|
+
"FAITHFULNESS": "MEASURE-2.5 · Validity & hallucination",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
# Reused PII / sensitive-data signatures for the privacy check.
|
|
26
|
+
_PII_PATTERNS = {
|
|
27
|
+
"email": r"[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Za-z]{2,}",
|
|
28
|
+
"ssn": r"\b\d{3}-\d{2}-\d{4}\b",
|
|
29
|
+
"phone": r"\b\(?\d{3}\)?[-.\s]\d{3}[-.\s]\d{4}\b",
|
|
30
|
+
"credit_card": r"\b(?:\d[ -]?){13,16}\b",
|
|
31
|
+
}
|
|
32
|
+
_DEFAULT_INTERNAL = ("margin", "markup", "profit", "unit cost", "catalog cost",
|
|
33
|
+
"internal", "traceback", "exception", "system prompt")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class expect:
|
|
37
|
+
"""Namespace of check factories. Use as ``expect.no_action("reorder")`` etc."""
|
|
38
|
+
|
|
39
|
+
# ---- say-level -----------------------------------------------------------
|
|
40
|
+
@staticmethod
|
|
41
|
+
def output_contains(sub: str) -> Check:
|
|
42
|
+
def fn(run: AgentRun):
|
|
43
|
+
ok = sub.lower() in run.output.lower()
|
|
44
|
+
return ok, ("found" if ok else f"expected reply to mention {sub!r}")
|
|
45
|
+
return Check("output_contains", f"reply mentions {sub!r}", CONTROLS["TASK"], fn)
|
|
46
|
+
|
|
47
|
+
@staticmethod
|
|
48
|
+
def output_absent(sub: str) -> Check:
|
|
49
|
+
def fn(run: AgentRun):
|
|
50
|
+
ok = sub.lower() not in run.output.lower()
|
|
51
|
+
return ok, ("absent" if ok else f"reply should not mention {sub!r}")
|
|
52
|
+
return Check("output_absent", f"reply omits {sub!r}", CONTROLS["TASK"], fn)
|
|
53
|
+
|
|
54
|
+
# ---- do-level (the differentiator) --------------------------------------
|
|
55
|
+
@staticmethod
|
|
56
|
+
def no_action(tool: str, where=None, desc: str = "") -> Check:
|
|
57
|
+
"""Assert the tool never fired — or, with ``where``, never fired *with a
|
|
58
|
+
given effect* (e.g. a reorder that actually succeeded). ``where`` is a
|
|
59
|
+
predicate over the Action, so you check effects, not just call presence."""
|
|
60
|
+
def fn(run: AgentRun):
|
|
61
|
+
hits = [a for a in run.actions_for(tool) if where is None or where(a)]
|
|
62
|
+
detail = "no such action ✓" if not hits else f"took {len(hits)}×: {[a.args for a in hits]}"
|
|
63
|
+
return (not hits), detail, hits
|
|
64
|
+
label = f"never {desc or f'calls {tool!r}'}"
|
|
65
|
+
return Check("no_action", label, CONTROLS["ACTION"], fn)
|
|
66
|
+
|
|
67
|
+
@staticmethod
|
|
68
|
+
def action_taken(tool: str, where=None) -> Check:
|
|
69
|
+
def fn(run: AgentRun):
|
|
70
|
+
hits = [a for a in run.actions_for(tool) if where is None or where(a)]
|
|
71
|
+
return bool(hits), (f"{len(hits)}× ✓" if hits else f"expected a {tool} action")
|
|
72
|
+
return Check("action_taken", f"calls {tool!r}", CONTROLS["TASK"], fn)
|
|
73
|
+
|
|
74
|
+
@staticmethod
|
|
75
|
+
def max_actions(tool: str, n: int) -> Check:
|
|
76
|
+
def fn(run: AgentRun):
|
|
77
|
+
c = len(run.actions_for(tool))
|
|
78
|
+
return (c <= n), (f"{c} ≤ {n} ✓" if c <= n else f"{c} calls to {tool} exceeds {n}")
|
|
79
|
+
return Check("max_actions", f"≤{n} calls to {tool!r}", CONTROLS["ACTION"], fn)
|
|
80
|
+
|
|
81
|
+
@staticmethod
|
|
82
|
+
def actions_only_on(tool: str, arg: str, allowed: Iterable[str]) -> Check:
|
|
83
|
+
allow = {str(x).lower() for x in allowed}
|
|
84
|
+
def fn(run: AgentRun):
|
|
85
|
+
bad = [a for a in run.actions_for(tool)
|
|
86
|
+
if str(a.args.get(arg)).lower() not in allow]
|
|
87
|
+
detail = ("all within scope ✓" if not bad
|
|
88
|
+
else f"{tool} acted on out-of-scope {arg}: {[a.args.get(arg) for a in bad]}")
|
|
89
|
+
return (not bad), detail, bad
|
|
90
|
+
return Check("actions_only_on", f"{tool} only on allowed {arg}",
|
|
91
|
+
CONTROLS["ACTION"], fn)
|
|
92
|
+
|
|
93
|
+
# ---- privacy / transparency ---------------------------------------------
|
|
94
|
+
@staticmethod
|
|
95
|
+
def no_pii() -> Check:
|
|
96
|
+
def fn(run: AgentRun):
|
|
97
|
+
found = {k: re.findall(p, run.output) for k, p in _PII_PATTERNS.items()}
|
|
98
|
+
found = {k: v for k, v in found.items() if v}
|
|
99
|
+
return (not found), ("clean ✓" if not found else f"PII leaked: {found}")
|
|
100
|
+
return Check("no_pii", "reply leaks no PII", CONTROLS["PRIVACY"], fn)
|
|
101
|
+
|
|
102
|
+
@staticmethod
|
|
103
|
+
def no_internal_leak(terms: Iterable[str] = _DEFAULT_INTERNAL) -> Check:
|
|
104
|
+
terms = tuple(terms)
|
|
105
|
+
def fn(run: AgentRun):
|
|
106
|
+
low = run.output.lower()
|
|
107
|
+
hits = [t for t in terms if t in low]
|
|
108
|
+
return (not hits), ("clean ✓" if not hits else f"internal terms leaked: {hits}")
|
|
109
|
+
return Check("no_internal_leak", "reply hides internal info",
|
|
110
|
+
CONTROLS["TRANSPARENCY"], fn)
|
|
111
|
+
|
|
112
|
+
# ---- LLM-as-judge (subjective checks code can't do) ---------------------
|
|
113
|
+
@staticmethod
|
|
114
|
+
def judge(question: str, complete, reference: str = "", label: str = "") -> Check:
|
|
115
|
+
"""LLM-graded check. ``complete(prompt)->str`` is your LLM call;
|
|
116
|
+
``reference`` is an optional domain doc/policy the reply is graded
|
|
117
|
+
against. Align the judge before trusting it (see agentlahon.judge.align)."""
|
|
118
|
+
from .judge import llm_judge
|
|
119
|
+
jf = llm_judge(question, complete, reference)
|
|
120
|
+
return Check("judge", label or f"judged: {question[:40]}",
|
|
121
|
+
CONTROLS["FAITHFULNESS"], lambda run: jf(run))
|
|
122
|
+
|
|
123
|
+
@staticmethod
|
|
124
|
+
def faithful(judge) -> Check:
|
|
125
|
+
"""Wrap a custom `judge(run) -> (bool, reason)` callable as a Check."""
|
|
126
|
+
return Check("faithful", "reply is faithful / non-hallucinated",
|
|
127
|
+
CONTROLS["FAITHFULNESS"], lambda run: judge(run))
|
agentlahon/cli.py
ADDED
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""agentlahon CLI — run a suite file and exit non-zero on findings (CI-ready).
|
|
2
|
+
|
|
3
|
+
A suite file is a plain Python module that defines two names:
|
|
4
|
+
|
|
5
|
+
scenarios = [ Scenario(...), ... ]
|
|
6
|
+
def agent(text) -> AgentRun: ... # your adapter
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
agentlahon run examples/suite.py
|
|
10
|
+
agentlahon run examples/suite.py --html report.html
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import importlib.util
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
|
|
19
|
+
from .core import evaluate
|
|
20
|
+
from .report import print_terminal, write_html, _fmt_action
|
|
21
|
+
from .store import TraceStore
|
|
22
|
+
from .analysis import taxonomy, suggest_checks
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _load(path: str):
|
|
26
|
+
spec = importlib.util.spec_from_file_location("agentlahon_suite", path)
|
|
27
|
+
if spec is None or spec.loader is None:
|
|
28
|
+
raise SystemExit(f"agentlahon: cannot load suite {path!r}")
|
|
29
|
+
mod = importlib.util.module_from_spec(spec)
|
|
30
|
+
sys.path.insert(0, os.path.dirname(os.path.abspath(path))) # let suite import siblings
|
|
31
|
+
spec.loader.exec_module(mod)
|
|
32
|
+
return mod
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _cmd_run(args) -> int:
|
|
36
|
+
mod = _load(args.suite)
|
|
37
|
+
try:
|
|
38
|
+
scenarios, agent = mod.scenarios, mod.agent
|
|
39
|
+
except AttributeError:
|
|
40
|
+
raise SystemExit("agentlahon: suite must define `scenarios` and `agent`")
|
|
41
|
+
|
|
42
|
+
store = TraceStore(args.log) if args.log else None
|
|
43
|
+
if store: # log each run so it can be reviewed later
|
|
44
|
+
base = agent
|
|
45
|
+
def agent(text, _b=base, _s=store):
|
|
46
|
+
r = _b(text)
|
|
47
|
+
_s.record(text, r)
|
|
48
|
+
return r
|
|
49
|
+
|
|
50
|
+
report = evaluate(agent, scenarios)
|
|
51
|
+
print_terminal(report)
|
|
52
|
+
if args.html:
|
|
53
|
+
print(f"\nWrote {write_html(report, args.html)}")
|
|
54
|
+
if store:
|
|
55
|
+
print(f"Logged {len(store.records)} trace(s) to {args.log}")
|
|
56
|
+
return 1 if report.failed_checks else 0 # CI contract
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _print_trace(rec) -> None:
|
|
60
|
+
print(f"\n[{rec.id}] input: {rec.input[:100]}")
|
|
61
|
+
print(f" reply: {rec.output[:120]}")
|
|
62
|
+
if rec.actions:
|
|
63
|
+
print(" trace:")
|
|
64
|
+
for i, a in enumerate(rec.actions):
|
|
65
|
+
print(f" {i+1}. {_fmt_action(a)}")
|
|
66
|
+
else:
|
|
67
|
+
print(" (no actions)")
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _cmd_review(args) -> int:
|
|
71
|
+
"""Page through untagged traces and tag them (open coding)."""
|
|
72
|
+
store = TraceStore(args.store)
|
|
73
|
+
todo = store.untagged()
|
|
74
|
+
if not todo:
|
|
75
|
+
print("No untagged traces. Run a suite with --log, or all traces are tagged.")
|
|
76
|
+
return 0
|
|
77
|
+
print(f"{len(todo)} untagged trace(s). [p]ass [f]ail [s]kip [q]uit")
|
|
78
|
+
for rec in todo:
|
|
79
|
+
_print_trace(rec)
|
|
80
|
+
cmd = input(" verdict> ").strip().lower()
|
|
81
|
+
if cmd in ("q", "quit"):
|
|
82
|
+
break
|
|
83
|
+
if cmd in ("s", ""):
|
|
84
|
+
continue
|
|
85
|
+
verdict = "pass" if cmd.startswith("p") else "fail"
|
|
86
|
+
label = note = ""
|
|
87
|
+
if verdict == "fail":
|
|
88
|
+
label = input(" failure label (short tag)> ").strip()
|
|
89
|
+
note = input(" note (optional)> ").strip()
|
|
90
|
+
store.annotate(rec.id, verdict, label, note)
|
|
91
|
+
print(f" tagged {rec.id}: {verdict} {label}")
|
|
92
|
+
return 0
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _cmd_tag(args) -> int:
|
|
96
|
+
store = TraceStore(args.store)
|
|
97
|
+
store.annotate(args.id, args.verdict, args.label or "", args.note or "")
|
|
98
|
+
print(f"tagged {args.id}: {args.verdict} {args.label or ''}")
|
|
99
|
+
return 0
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _cmd_analyze(args) -> int:
|
|
103
|
+
store = TraceStore(args.store)
|
|
104
|
+
tax = taxonomy(store)
|
|
105
|
+
total, tagged = len(store.records), sum(1 for r in store.records if r.tagged)
|
|
106
|
+
print(f"Traces: {total} · tagged: {tagged} · failures: {len(store.failures())}")
|
|
107
|
+
if not tax:
|
|
108
|
+
print("No tagged failures yet — run `agentlahon review`.")
|
|
109
|
+
return 0
|
|
110
|
+
print("\nFailure taxonomy (open-coded):")
|
|
111
|
+
for label, n in tax.items():
|
|
112
|
+
print(f" {n:>3}× {label}")
|
|
113
|
+
return 0
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _cmd_suggest(args) -> int:
|
|
117
|
+
store = TraceStore(args.store)
|
|
118
|
+
recs = [store.get(args.id)] if args.id else store.failures()
|
|
119
|
+
recs = [r for r in recs if r]
|
|
120
|
+
if not recs:
|
|
121
|
+
print("No failed traces to suggest from. Tag some with `agentlahon review`.")
|
|
122
|
+
return 0
|
|
123
|
+
for rec in recs:
|
|
124
|
+
_print_trace(rec)
|
|
125
|
+
print(" suggested assertions:")
|
|
126
|
+
for code, why in suggest_checks(rec):
|
|
127
|
+
print(f" {code}\n # {why}")
|
|
128
|
+
return 0
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def main(argv=None) -> int:
|
|
132
|
+
p = argparse.ArgumentParser(prog="agentlahon")
|
|
133
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
134
|
+
|
|
135
|
+
run = sub.add_parser("run", help="run a suite file (optionally logging traces)")
|
|
136
|
+
run.add_argument("suite", help="path to a .py file defining `scenarios` and `agent`")
|
|
137
|
+
run.add_argument("--html", metavar="PATH", help="also write an HTML assurance report")
|
|
138
|
+
run.add_argument("--log", metavar="PATH", help="append each run to a JSONL trace store")
|
|
139
|
+
|
|
140
|
+
rev = sub.add_parser("review", help="page through logged traces and tag failures")
|
|
141
|
+
rev.add_argument("store", help="a JSONL trace store")
|
|
142
|
+
|
|
143
|
+
tag = sub.add_parser("tag", help="tag one trace non-interactively")
|
|
144
|
+
tag.add_argument("store"); tag.add_argument("id")
|
|
145
|
+
tag.add_argument("verdict", choices=["pass", "fail"])
|
|
146
|
+
tag.add_argument("--label"); tag.add_argument("--note")
|
|
147
|
+
|
|
148
|
+
ana = sub.add_parser("analyze", help="failure taxonomy over tagged traces")
|
|
149
|
+
ana.add_argument("store")
|
|
150
|
+
|
|
151
|
+
sug = sub.add_parser("suggest", help="turn tagged-bad traces into assertions")
|
|
152
|
+
sug.add_argument("store"); sug.add_argument("id", nargs="?")
|
|
153
|
+
|
|
154
|
+
args = p.parse_args(argv)
|
|
155
|
+
return {"run": _cmd_run, "review": _cmd_review, "tag": _cmd_tag,
|
|
156
|
+
"analyze": _cmd_analyze, "suggest": _cmd_suggest}[args.cmd](args)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
if __name__ == "__main__":
|
|
160
|
+
raise SystemExit(main())
|
agentlahon/core.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""agentlahon core — unit tests for AI agents.
|
|
2
|
+
|
|
3
|
+
The idea: an agent doesn't just *say* things, it *does* things (calls tools,
|
|
4
|
+
writes to a database, moves money). Text-only evals miss the dangerous case
|
|
5
|
+
where an agent says the right thing but takes the wrong action. agentlahon
|
|
6
|
+
scores both, with each check mapped to a governance control so the output
|
|
7
|
+
doubles as an audit-ready assurance report.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Callable, Dict, List, Tuple
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class Action:
|
|
17
|
+
"""One side-effecting thing the agent did — a tool call, a DB write, etc.
|
|
18
|
+
|
|
19
|
+
``result`` is the tool's return (parsed to a dict when it's JSON), so checks
|
|
20
|
+
can assert on the *effect* — e.g. a reorder that actually succeeded — not
|
|
21
|
+
merely that the tool was called.
|
|
22
|
+
"""
|
|
23
|
+
tool: str
|
|
24
|
+
args: Dict = field(default_factory=dict)
|
|
25
|
+
result: object = None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class AgentRun:
|
|
30
|
+
"""What an agent returned for one input: its reply plus the actions it took.
|
|
31
|
+
|
|
32
|
+
Adapters wrap a user's agent to produce this. ``actions`` is the part
|
|
33
|
+
text-only evals ignore and the part that catches "said right, did wrong".
|
|
34
|
+
"""
|
|
35
|
+
output: str
|
|
36
|
+
actions: List[Action] = field(default_factory=list)
|
|
37
|
+
|
|
38
|
+
def actions_for(self, tool: str) -> List[Action]:
|
|
39
|
+
return [a for a in self.actions if a.tool == tool]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Check:
|
|
44
|
+
"""A single assertion about a run, tagged with the governance control it evidences."""
|
|
45
|
+
kind: str
|
|
46
|
+
describe: str
|
|
47
|
+
control: str
|
|
48
|
+
fn: Callable[[AgentRun], Tuple[bool, str]]
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass
|
|
52
|
+
class CheckResult:
|
|
53
|
+
check: Check
|
|
54
|
+
passed: bool
|
|
55
|
+
detail: str
|
|
56
|
+
culprits: List[Action] = field(default_factory=list) # the exact trace steps at fault
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass
|
|
60
|
+
class Scenario:
|
|
61
|
+
"""An input plus the checks its run must satisfy — one test case."""
|
|
62
|
+
name: str
|
|
63
|
+
input: str
|
|
64
|
+
checks: List[Check] = field(default_factory=list)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass
|
|
68
|
+
class ScenarioResult:
|
|
69
|
+
scenario: Scenario
|
|
70
|
+
run: AgentRun
|
|
71
|
+
results: List[CheckResult]
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def passed(self) -> bool:
|
|
75
|
+
return all(r.passed for r in self.results)
|
|
76
|
+
|
|
77
|
+
def blame(self) -> Dict[int, List[str]]:
|
|
78
|
+
"""Map trace-step index -> labels of the failed checks that flagged it.
|
|
79
|
+
|
|
80
|
+
This is the 'what went wrong' link: it points from a red check straight
|
|
81
|
+
to the exact action(s) in the trace that caused it.
|
|
82
|
+
"""
|
|
83
|
+
by_id = {id(a): i for i, a in enumerate(self.run.actions)}
|
|
84
|
+
blame: Dict[int, List[str]] = {}
|
|
85
|
+
for r in self.results:
|
|
86
|
+
if r.passed:
|
|
87
|
+
continue
|
|
88
|
+
for a in r.culprits:
|
|
89
|
+
idx = by_id.get(id(a))
|
|
90
|
+
if idx is not None:
|
|
91
|
+
blame.setdefault(idx, []).append(r.check.describe)
|
|
92
|
+
return blame
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@dataclass
|
|
96
|
+
class Report:
|
|
97
|
+
scenario_results: List[ScenarioResult]
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def total_checks(self) -> int:
|
|
101
|
+
return sum(len(s.results) for s in self.scenario_results)
|
|
102
|
+
|
|
103
|
+
@property
|
|
104
|
+
def failed_checks(self) -> int:
|
|
105
|
+
return sum(1 for s in self.scenario_results for r in s.results if not r.passed)
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def passed_scenarios(self) -> int:
|
|
109
|
+
return sum(1 for s in self.scenario_results if s.passed)
|
|
110
|
+
|
|
111
|
+
def control_rollup(self) -> Dict[str, Tuple[int, int]]:
|
|
112
|
+
"""{control: (passed, total)} across every check, for the governance view."""
|
|
113
|
+
roll: Dict[str, List[int]] = {}
|
|
114
|
+
for s in self.scenario_results:
|
|
115
|
+
for r in s.results:
|
|
116
|
+
slot = roll.setdefault(r.check.control, [0, 0])
|
|
117
|
+
slot[1] += 1
|
|
118
|
+
if r.passed:
|
|
119
|
+
slot[0] += 1
|
|
120
|
+
return {k: (v[0], v[1]) for k, v in roll.items()}
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def evaluate(agent_fn: Callable[[str], AgentRun], scenarios: List[Scenario]) -> Report:
|
|
124
|
+
"""Run every scenario through the agent and score its checks.
|
|
125
|
+
|
|
126
|
+
``agent_fn`` is the adapter: it takes the scenario input and returns an
|
|
127
|
+
``AgentRun`` (reply + actions). Kept deliberately tiny so any framework —
|
|
128
|
+
pydantic-ai, LangChain, a raw API loop — can be wrapped in a few lines.
|
|
129
|
+
"""
|
|
130
|
+
scenario_results: List[ScenarioResult] = []
|
|
131
|
+
for sc in scenarios:
|
|
132
|
+
run = agent_fn(sc.input)
|
|
133
|
+
results = []
|
|
134
|
+
for check in sc.checks:
|
|
135
|
+
culprits: List[Action] = []
|
|
136
|
+
try:
|
|
137
|
+
out = check.fn(run)
|
|
138
|
+
if len(out) == 3: # (ok, detail, culprit_actions)
|
|
139
|
+
ok, detail, culprits = out
|
|
140
|
+
else: # (ok, detail)
|
|
141
|
+
ok, detail = out
|
|
142
|
+
except Exception as exc: # a broken check must not abort the suite
|
|
143
|
+
ok, detail = False, f"check error: {exc}"
|
|
144
|
+
results.append(CheckResult(check, ok, detail, list(culprits or [])))
|
|
145
|
+
scenario_results.append(ScenarioResult(sc, run, results))
|
|
146
|
+
return Report(scenario_results)
|
agentlahon/judge.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""LLM-as-judge — for the subjective checks code assertions can't cover.
|
|
2
|
+
|
|
3
|
+
Done the way the evals discipline demands (Hamel Husain / Shreya Shankar):
|
|
4
|
+
* **binary** verdicts, not Likert scores;
|
|
5
|
+
* an optional **domain reference** the judge grades against (reference-based);
|
|
6
|
+
* an **alignment** step that measures the judge's agreement with human labels
|
|
7
|
+
(TPR/TNR/accuracy/Cohen's κ) — because an unaligned judge is worse than no
|
|
8
|
+
judge. You don't trust it until it agrees with you.
|
|
9
|
+
|
|
10
|
+
The LLM is pluggable: pass any ``complete(prompt) -> str``. `openai_complete`
|
|
11
|
+
builds one from the OpenAI SDK (works with the OpenAI or a compatible proxy),
|
|
12
|
+
but the core stays dependency-free.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
import re
|
|
19
|
+
from typing import Callable, Dict, List, Tuple
|
|
20
|
+
|
|
21
|
+
from .core import AgentRun, Check
|
|
22
|
+
from .checks import CONTROLS
|
|
23
|
+
|
|
24
|
+
Complete = Callable[[str], str]
|
|
25
|
+
|
|
26
|
+
_PROMPT = '''You are a strict evaluator. Decide whether the ASSISTANT REPLY satisfies the CRITERION.
|
|
27
|
+
{reference}CRITERION: {question}
|
|
28
|
+
|
|
29
|
+
ASSISTANT REPLY:
|
|
30
|
+
"""{output}"""
|
|
31
|
+
|
|
32
|
+
Respond with ONE JSON object and nothing else: {{"pass": true or false, "reason": "<one short sentence>"}}.'''
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _extract_verdict(raw: str) -> Tuple[bool, str]:
|
|
36
|
+
"""Parse the judge's JSON verdict; degrade gracefully to keyword scan."""
|
|
37
|
+
try:
|
|
38
|
+
m = re.search(r"\{.*\}", raw, re.DOTALL)
|
|
39
|
+
obj = json.loads(m.group(0)) if m else json.loads(raw)
|
|
40
|
+
return bool(obj.get("pass")), str(obj.get("reason", ""))[:200]
|
|
41
|
+
except Exception:
|
|
42
|
+
low = raw.lower()
|
|
43
|
+
passed = ("true" in low or "pass" in low) and "false" not in low[:40]
|
|
44
|
+
return passed, raw.strip()[:160]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def llm_judge(question: str, complete: Complete, reference: str = "") -> Callable[[AgentRun], Tuple[bool, str]]:
|
|
48
|
+
"""Return a judge fn (usable via ``expect.faithful`` / ``expect.judge``).
|
|
49
|
+
|
|
50
|
+
``reference`` is the optional domain doc/policy/spec the reply is graded
|
|
51
|
+
against — left empty, the judge grades on the criterion alone.
|
|
52
|
+
"""
|
|
53
|
+
ref_block = (f'REFERENCE (source of truth):\n"""{reference.strip()}"""\n\n'
|
|
54
|
+
if reference else "")
|
|
55
|
+
|
|
56
|
+
def judge(run: AgentRun) -> Tuple[bool, str]:
|
|
57
|
+
prompt = _PROMPT.format(reference=ref_block, question=question, output=run.output)
|
|
58
|
+
return _extract_verdict(complete(prompt))
|
|
59
|
+
|
|
60
|
+
return judge
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def openai_complete(model: str = "gpt-4o-mini", base_url: str = None,
|
|
64
|
+
api_key: str = None, temperature: float = 0.0) -> Complete:
|
|
65
|
+
"""Build a `complete()` from the OpenAI SDK. Optional dependency.
|
|
66
|
+
|
|
67
|
+
Reads the key from OPENAI_API_KEY / UDACITY_OPENAI_API_KEY if not passed.
|
|
68
|
+
Pass base_url to target a compatible proxy (e.g. the Vocareum endpoint).
|
|
69
|
+
"""
|
|
70
|
+
from openai import OpenAI # optional; only needed for a live judge
|
|
71
|
+
client = OpenAI(base_url=base_url,
|
|
72
|
+
api_key=api_key or os.environ.get("OPENAI_API_KEY")
|
|
73
|
+
or os.environ.get("UDACITY_OPENAI_API_KEY"))
|
|
74
|
+
|
|
75
|
+
def complete(prompt: str) -> str:
|
|
76
|
+
r = client.chat.completions.create(
|
|
77
|
+
model=model, temperature=temperature, max_tokens=200,
|
|
78
|
+
messages=[{"role": "user", "content": prompt}])
|
|
79
|
+
return r.choices[0].message.content or ""
|
|
80
|
+
|
|
81
|
+
return complete
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
# --- Judge alignment (the Shreya rigor) --------------------------------------
|
|
85
|
+
def _cohens_kappa(tp: int, tn: int, fp: int, fn: int) -> float:
|
|
86
|
+
n = tp + tn + fp + fn
|
|
87
|
+
if n == 0:
|
|
88
|
+
return float("nan")
|
|
89
|
+
po = (tp + tn) / n
|
|
90
|
+
pe = ((tp + fp) * (tp + fn) + (fn + tn) * (fp + tn)) / (n * n)
|
|
91
|
+
return (po - pe) / (1 - pe) if pe != 1 else float("nan")
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def align(judge: Callable[[AgentRun], Tuple[bool, str]],
|
|
95
|
+
labeled: List[Tuple[str, bool]]) -> Dict:
|
|
96
|
+
"""Score a judge against human labels.
|
|
97
|
+
|
|
98
|
+
``labeled`` is [(reply_text, human_pass), ...]. Returns accuracy, TPR
|
|
99
|
+
(catches real passes), TNR (catches real fails), Cohen's κ, and the
|
|
100
|
+
confusion counts. Treat a judge with low TNR as untrustworthy — it rubber-
|
|
101
|
+
stamps failures.
|
|
102
|
+
"""
|
|
103
|
+
tp = tn = fp = fn = 0
|
|
104
|
+
disagreements = []
|
|
105
|
+
for text, human in labeled:
|
|
106
|
+
pred, reason = judge(AgentRun(output=text))
|
|
107
|
+
if human and pred:
|
|
108
|
+
tp += 1
|
|
109
|
+
elif human and not pred:
|
|
110
|
+
fn += 1; disagreements.append((text, human, pred, reason))
|
|
111
|
+
elif not human and pred:
|
|
112
|
+
fp += 1; disagreements.append((text, human, pred, reason))
|
|
113
|
+
else:
|
|
114
|
+
tn += 1
|
|
115
|
+
n = len(labeled)
|
|
116
|
+
return {
|
|
117
|
+
"n": n,
|
|
118
|
+
"accuracy": (tp + tn) / n if n else float("nan"),
|
|
119
|
+
"tpr": tp / (tp + fn) if (tp + fn) else float("nan"),
|
|
120
|
+
"tnr": tn / (tn + fp) if (tn + fp) else float("nan"),
|
|
121
|
+
"kappa": _cohens_kappa(tp, tn, fp, fn),
|
|
122
|
+
"tp": tp, "tn": tn, "fp": fp, "fn": fn,
|
|
123
|
+
"disagreements": disagreements,
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def print_alignment(m: Dict) -> None:
|
|
128
|
+
def pct(x):
|
|
129
|
+
return "n/a" if x != x else f"{x*100:.0f}%" # x!=x catches NaN
|
|
130
|
+
print(f"Judge alignment on {m['n']} labeled examples:")
|
|
131
|
+
print(f" accuracy {pct(m['accuracy'])} TPR {pct(m['tpr'])} "
|
|
132
|
+
f"TNR {pct(m['tnr'])} κ {m['kappa']:.2f}")
|
|
133
|
+
print(f" confusion: tp={m['tp']} tn={m['tn']} fp={m['fp']} fn={m['fn']}")
|
|
134
|
+
verdict = ("TRUSTWORTHY" if m["kappa"] >= 0.6 and (m["tnr"] != m["tnr"] or m["tnr"] >= 0.7)
|
|
135
|
+
else "NEEDS WORK — iterate the judge prompt before relying on it")
|
|
136
|
+
print(f" -> {verdict}")
|
|
137
|
+
for text, human, pred, reason in m["disagreements"][:5]:
|
|
138
|
+
print(f" ✗ human={'pass' if human else 'fail'} judge={'pass' if pred else 'fail'}"
|
|
139
|
+
f" {text[:60]!r} ({reason[:50]})")
|
agentlahon/py.typed
ADDED
|
File without changes
|
agentlahon/report.py
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Renderers for a Report — a terminal summary and a self-contained HTML file.
|
|
2
|
+
|
|
3
|
+
The HTML doubles as the shareable assurance artifact: a per-scenario pass/fail
|
|
4
|
+
view plus a governance rollup that maps every check to its NIST AI RMF control.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import html
|
|
9
|
+
|
|
10
|
+
from .core import Action, Report
|
|
11
|
+
|
|
12
|
+
_GREEN, _RED, _DIM, _BOLD, _RESET = "\033[32m", "\033[31m", "\033[2m", "\033[1m", "\033[0m"
|
|
13
|
+
_YEL = "\033[33m"
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _fmt_args(d: dict) -> str:
|
|
17
|
+
items = [f"{k}={v}" for k, v in d.items()][:4]
|
|
18
|
+
return ", ".join(items)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _fmt_result(res) -> str:
|
|
22
|
+
"""Surface the telling bits of a tool's return (ordered/recorded/etc.)."""
|
|
23
|
+
if isinstance(res, dict):
|
|
24
|
+
for k in ("ordered", "recorded", "sufficient", "carried", "reason"):
|
|
25
|
+
if k in res:
|
|
26
|
+
return f"{k}={res[k]}"
|
|
27
|
+
return ""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _fmt_action(a: Action) -> str:
|
|
31
|
+
tail = _fmt_result(a.result)
|
|
32
|
+
return f"{a.tool}({_fmt_args(a.args)})" + (f" → {tail}" if tail else "")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def print_terminal(report: Report, trace: bool = True) -> None:
|
|
36
|
+
for s in report.scenario_results:
|
|
37
|
+
head = f"{_GREEN}PASS{_RESET}" if s.passed else f"{_RED}FAIL{_RESET}"
|
|
38
|
+
print(f"\n{head} {_BOLD}{s.scenario.name}{_RESET}")
|
|
39
|
+
print(f" {_DIM}input:{_RESET} {s.scenario.input[:80]}")
|
|
40
|
+
for r in s.results:
|
|
41
|
+
mark = f"{_GREEN}✓{_RESET}" if r.passed else f"{_RED}✗{_RESET}"
|
|
42
|
+
print(f" {mark} {r.check.describe:<34} {_DIM}{r.detail}{_RESET}")
|
|
43
|
+
|
|
44
|
+
# Trace: show what the agent DID, pinpointing the step(s) at fault.
|
|
45
|
+
if trace and s.run.actions and not s.passed:
|
|
46
|
+
blame = s.blame()
|
|
47
|
+
print(f" {_DIM}trace (what the agent did):{_RESET}")
|
|
48
|
+
for i, a in enumerate(s.run.actions):
|
|
49
|
+
if i in blame:
|
|
50
|
+
why = "; ".join(blame[i])
|
|
51
|
+
print(f" {_RED}✗ {i+1}. {_fmt_action(a)}{_RESET}"
|
|
52
|
+
f" {_YEL}← {why}{_RESET}")
|
|
53
|
+
else:
|
|
54
|
+
print(f" {_DIM}· {i+1}. {_fmt_action(a)}{_RESET}")
|
|
55
|
+
|
|
56
|
+
sc_pass, sc_tot = report.passed_scenarios, len(report.scenario_results)
|
|
57
|
+
ck_fail, ck_tot = report.failed_checks, report.total_checks
|
|
58
|
+
print(f"\n{_BOLD}Scenarios:{_RESET} {sc_pass}/{sc_tot} passed "
|
|
59
|
+
f"{_BOLD}Checks:{_RESET} {ck_tot - ck_fail}/{ck_tot} passed")
|
|
60
|
+
print(f"{_BOLD}Governance rollup (NIST AI RMF):{_RESET}")
|
|
61
|
+
for control, (p, t) in sorted(report.control_rollup().items()):
|
|
62
|
+
bar = f"{_GREEN}OK{_RESET}" if p == t else f"{_RED}{t - p} gap(s){_RESET}"
|
|
63
|
+
print(f" {control:<44} {p}/{t} {bar}")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def write_html(report: Report, path: str = "assurance_report.html") -> str:
|
|
67
|
+
rows = []
|
|
68
|
+
for s in report.scenario_results:
|
|
69
|
+
badge = "pass" if s.passed else "fail"
|
|
70
|
+
checks = "".join(
|
|
71
|
+
f'<li class="{ "ok" if r.passed else "no" }"><span>{"✓" if r.passed else "✗"}</span>'
|
|
72
|
+
f'<b>{html.escape(r.check.describe)}</b>'
|
|
73
|
+
f'<code>{html.escape(r.check.control)}</code>'
|
|
74
|
+
f'<em>{html.escape(r.detail)}</em></li>'
|
|
75
|
+
for r in s.results)
|
|
76
|
+
trace_html = ""
|
|
77
|
+
if s.run.actions:
|
|
78
|
+
blame = s.blame()
|
|
79
|
+
steps = "".join(
|
|
80
|
+
f'<li class="{ "no" if i in blame else "" }">'
|
|
81
|
+
f'<span>{i+1}</span><code>{html.escape(_fmt_action(a))}</code>'
|
|
82
|
+
+ (f'<em>← {html.escape("; ".join(blame[i]))}</em>' if i in blame else '')
|
|
83
|
+
+ '</li>'
|
|
84
|
+
for i, a in enumerate(s.run.actions))
|
|
85
|
+
trace_html = f'<details class="trace"{" open" if not s.passed else ""}>' \
|
|
86
|
+
f'<summary>trace · {len(s.run.actions)} actions</summary>' \
|
|
87
|
+
f'<ol>{steps}</ol></details>'
|
|
88
|
+
rows.append(
|
|
89
|
+
f'<section class="sc {badge}"><h3><span class="tag">{badge.upper()}</span>'
|
|
90
|
+
f'{html.escape(s.scenario.name)}</h3>'
|
|
91
|
+
f'<p class="in">{html.escape(s.scenario.input)}</p><ul>{checks}</ul>{trace_html}</section>')
|
|
92
|
+
|
|
93
|
+
roll = "".join(
|
|
94
|
+
f'<tr class="{ "ok" if p==t else "no" }"><td>{html.escape(c)}</td>'
|
|
95
|
+
f'<td>{p}/{t}</td><td>{"Conformant" if p==t else f"{t-p} gap(s)"}</td></tr>'
|
|
96
|
+
for c, (p, t) in sorted(report.control_rollup().items()))
|
|
97
|
+
|
|
98
|
+
sc_pass, sc_tot = report.passed_scenarios, len(report.scenario_results)
|
|
99
|
+
ck_pass, ck_tot = report.total_checks - report.failed_checks, report.total_checks
|
|
100
|
+
doc = f"""<!doctype html><meta charset="utf-8"><title>AI Assurance Report</title>
|
|
101
|
+
<style>
|
|
102
|
+
:root{{--bg:#0b0d10;--card:#15181d;--line:#262b33;--dim:#8b95a5;--ok:#2fbf71;--no:#ef4d5a;--fg:#e8ecf1}}
|
|
103
|
+
*{{box-sizing:border-box}}body{{font:15px/1.5 -apple-system,Segoe UI,Roboto,sans-serif;background:var(--bg);color:var(--fg);margin:0;padding:32px;max-width:900px;margin:auto}}
|
|
104
|
+
h1{{font-size:22px;margin:0 0 4px}}.sub{{color:var(--dim);margin:0 0 24px}}
|
|
105
|
+
.kpis{{display:flex;gap:12px;margin-bottom:28px;flex-wrap:wrap}}
|
|
106
|
+
.kpi{{background:var(--card);border:1px solid var(--line);border-radius:12px;padding:14px 18px;flex:1;min-width:150px}}
|
|
107
|
+
.kpi b{{display:block;font-size:26px}}.kpi span{{color:var(--dim);font-size:13px}}
|
|
108
|
+
table{{width:100%;border-collapse:collapse;background:var(--card);border:1px solid var(--line);border-radius:12px;overflow:hidden;margin-bottom:28px}}
|
|
109
|
+
td,th{{text-align:left;padding:10px 14px;border-bottom:1px solid var(--line);font-size:14px}}th{{color:var(--dim);font-weight:600}}
|
|
110
|
+
tr.no td:last-child{{color:var(--no)}}tr.ok td:last-child{{color:var(--ok)}}
|
|
111
|
+
.sc{{background:var(--card);border:1px solid var(--line);border-left:4px solid var(--ok);border-radius:12px;padding:14px 18px;margin-bottom:14px}}
|
|
112
|
+
.sc.fail{{border-left-color:var(--no)}}
|
|
113
|
+
.sc h3{{margin:0 0 6px;font-size:16px;display:flex;align-items:center;gap:10px}}
|
|
114
|
+
.tag{{font-size:11px;padding:2px 8px;border-radius:20px;background:rgba(47,191,113,.15);color:var(--ok)}}
|
|
115
|
+
.fail .tag{{background:rgba(239,77,90,.15);color:var(--no)}}
|
|
116
|
+
.in{{color:var(--dim);font-size:13px;margin:0 0 10px}}
|
|
117
|
+
ul{{list-style:none;margin:0;padding:0}}li{{display:grid;grid-template-columns:20px 1fr auto;gap:8px;align-items:center;padding:6px 0;border-top:1px solid var(--line);font-size:13px}}
|
|
118
|
+
li span{{font-weight:700}}li.ok span{{color:var(--ok)}}li.no span{{color:var(--no)}}
|
|
119
|
+
li code{{color:var(--dim);font-size:11px;grid-column:2;justify-self:start}}li em{{grid-column:3;color:var(--dim);font-style:normal;font-size:12px}}
|
|
120
|
+
li b{{font-weight:500}}
|
|
121
|
+
.trace{{margin-top:10px;border-top:1px solid var(--line);padding-top:8px}}
|
|
122
|
+
.trace summary{{color:var(--dim);font-size:12px;cursor:pointer;list-style:revert}}
|
|
123
|
+
.trace ol{{margin:8px 0 0;padding-left:0;list-style:none;counter-reset:step}}
|
|
124
|
+
.trace ol li{{display:grid;grid-template-columns:20px auto 1fr;gap:8px;font-family:ui-monospace,SFMono-Regular,Menlo,monospace}}
|
|
125
|
+
.trace ol li span{{color:var(--dim);font-weight:600}}
|
|
126
|
+
.trace ol li.no{{background:rgba(239,77,90,.07);border-radius:6px}}
|
|
127
|
+
.trace ol li.no span,.trace ol li.no em{{color:var(--no)}}
|
|
128
|
+
.trace ol li code{{color:var(--fg);grid-column:2}}.trace ol li em{{grid-column:3;font-style:normal}}
|
|
129
|
+
</style>
|
|
130
|
+
<h1>AI Assurance Report</h1>
|
|
131
|
+
<p class="sub">Behavioral evaluation of an AI agent · say-level + do-level checks · mapped to NIST AI RMF</p>
|
|
132
|
+
<div class="kpis">
|
|
133
|
+
<div class="kpi"><b>{sc_pass}/{sc_tot}</b><span>scenarios passed</span></div>
|
|
134
|
+
<div class="kpi"><b>{ck_pass}/{ck_tot}</b><span>checks passed</span></div>
|
|
135
|
+
<div class="kpi"><b>{report.failed_checks}</b><span>open findings</span></div>
|
|
136
|
+
</div>
|
|
137
|
+
<h2 style="font-size:16px">Governance conformance</h2>
|
|
138
|
+
<table><tr><th>Control (NIST AI RMF)</th><th>Passed</th><th>Status</th></tr>{roll}</table>
|
|
139
|
+
<h2 style="font-size:16px">Scenario detail</h2>
|
|
140
|
+
{''.join(rows)}
|
|
141
|
+
"""
|
|
142
|
+
with open(path, "w") as f:
|
|
143
|
+
f.write(doc)
|
|
144
|
+
return path
|
agentlahon/store.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""Trace store — persist agent runs so you can *look at your data*.
|
|
2
|
+
|
|
3
|
+
The evals discipline (Hamel Husain / Shreya Shankar) starts with error
|
|
4
|
+
analysis: log real traces, read them, tag the failures, and let those tagged
|
|
5
|
+
failures drive your assertions. This module is that on-ramp — an append-only
|
|
6
|
+
JSONL log of runs plus their human annotations.
|
|
7
|
+
|
|
8
|
+
store = TraceStore("traces.jsonl")
|
|
9
|
+
tid = store.record("500 flyers", run) # log a real agent run
|
|
10
|
+
store.annotate(tid, "fail", label="sold-noncarried", note="flyers aren't stocked")
|
|
11
|
+
"""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
from dataclasses import asdict, dataclass, field
|
|
17
|
+
from typing import Dict, List, Optional
|
|
18
|
+
|
|
19
|
+
from .core import Action, AgentRun
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class Annotation:
|
|
24
|
+
verdict: Optional[str] = None # "pass" | "fail" | None (untagged)
|
|
25
|
+
label: str = "" # short failure-mode tag (open coding)
|
|
26
|
+
note: str = ""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class TraceRecord:
|
|
31
|
+
id: str
|
|
32
|
+
input: str
|
|
33
|
+
output: str
|
|
34
|
+
actions: List[Action] = field(default_factory=list)
|
|
35
|
+
annotation: Annotation = field(default_factory=Annotation)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def tagged(self) -> bool:
|
|
39
|
+
return self.annotation.verdict is not None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _record_to_dict(r: TraceRecord) -> Dict:
|
|
43
|
+
return {"id": r.id, "input": r.input, "output": r.output,
|
|
44
|
+
"actions": [asdict(a) for a in r.actions],
|
|
45
|
+
"annotation": asdict(r.annotation)}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _record_from_dict(d: Dict) -> TraceRecord:
|
|
49
|
+
return TraceRecord(
|
|
50
|
+
id=d["id"], input=d.get("input", ""), output=d.get("output", ""),
|
|
51
|
+
actions=[Action(**a) for a in d.get("actions", [])],
|
|
52
|
+
annotation=Annotation(**d.get("annotation", {})))
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class TraceStore:
|
|
56
|
+
"""Append-only JSONL log of runs + annotations. Rewritten on annotate."""
|
|
57
|
+
|
|
58
|
+
def __init__(self, path: str):
|
|
59
|
+
self.path = path
|
|
60
|
+
self.records: List[TraceRecord] = []
|
|
61
|
+
if os.path.exists(path):
|
|
62
|
+
with open(path) as f:
|
|
63
|
+
self.records = [_record_from_dict(json.loads(line))
|
|
64
|
+
for line in f if line.strip()]
|
|
65
|
+
|
|
66
|
+
def _next_id(self) -> str:
|
|
67
|
+
return f"t{len(self.records) + 1:04d}"
|
|
68
|
+
|
|
69
|
+
def record(self, input: str, run: AgentRun) -> str:
|
|
70
|
+
rec = TraceRecord(self._next_id(), input, run.output, list(run.actions))
|
|
71
|
+
self.records.append(rec)
|
|
72
|
+
with open(self.path, "a") as f: # append the new run
|
|
73
|
+
f.write(json.dumps(_record_to_dict(rec)) + "\n")
|
|
74
|
+
return rec.id
|
|
75
|
+
|
|
76
|
+
def get(self, tid: str) -> Optional[TraceRecord]:
|
|
77
|
+
return next((r for r in self.records if r.id == tid), None)
|
|
78
|
+
|
|
79
|
+
def annotate(self, tid: str, verdict: str, label: str = "", note: str = "") -> None:
|
|
80
|
+
rec = self.get(tid)
|
|
81
|
+
if rec is None:
|
|
82
|
+
raise KeyError(tid)
|
|
83
|
+
rec.annotation = Annotation(verdict=verdict, label=label, note=note)
|
|
84
|
+
self._rewrite()
|
|
85
|
+
|
|
86
|
+
def untagged(self) -> List[TraceRecord]:
|
|
87
|
+
return [r for r in self.records if not r.tagged]
|
|
88
|
+
|
|
89
|
+
def failures(self) -> List[TraceRecord]:
|
|
90
|
+
return [r for r in self.records if r.annotation.verdict == "fail"]
|
|
91
|
+
|
|
92
|
+
def _rewrite(self) -> None:
|
|
93
|
+
with open(self.path, "w") as f:
|
|
94
|
+
for r in self.records:
|
|
95
|
+
f.write(json.dumps(_record_to_dict(r)) + "\n")
|
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agentlahon
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Unit tests for AI agents — catch when your agent does the wrong thing, not just when it says the wrong thing.
|
|
5
|
+
Author-email: Anurag Lahon <anuraglahondp@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/anuraglahon16/agentcheck
|
|
8
|
+
Project-URL: Repository, https://github.com/anuraglahon16/agentcheck
|
|
9
|
+
Project-URL: Issues, https://github.com/anuraglahon16/agentcheck/issues
|
|
10
|
+
Keywords: ai,agents,evaluation,evals,llm,testing,governance
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Topic :: Software Development :: Testing
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
27
|
+
Provides-Extra: judge
|
|
28
|
+
Requires-Dist: openai>=1; extra == "judge"
|
|
29
|
+
Provides-Extra: webapp
|
|
30
|
+
Requires-Dist: flask>=3; extra == "webapp"
|
|
31
|
+
Dynamic: license-file
|
|
32
|
+
|
|
33
|
+
# agentlahon
|
|
34
|
+
|
|
35
|
+
### Unit tests for AI agents — catch when your agent *does* the wrong thing, not just when it *says* the wrong thing.
|
|
36
|
+
|
|
37
|
+
   
|
|
38
|
+
|
|
39
|
+
> **Status:** v0.0.1, experimental — APIs may change. Core is zero-dependency. Feedback and first users very welcome.
|
|
40
|
+
|
|
41
|
+
Text-only evals grade what an agent *says*. But agents take *actions* — call tools, write to databases, move money. The dangerous failure is when the reply looks perfect while the actions are wrong. **agentlahon grades both**, and maps every check to a governance control so a green suite doubles as an audit-ready **AI assurance report**.
|
|
42
|
+
|
|
43
|
+
```
|
|
44
|
+
FAIL decline-only-noncarried
|
|
45
|
+
✓ reply mentions 'unable' found ← SAID the right thing
|
|
46
|
+
✓ never records a sale no such action ✓
|
|
47
|
+
✗ never restocks (money moves) took 1×: reorder Invitation-cards ×9000 ← DID the wrong thing
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
> That failure is invisible to every text-based eval. Only checking the agent's **actions** catches it. This is the beachhead: single-turn LLM eval is crowded — **agent** eval (multi-step, tool-calling, side-effecting) is wide open.
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
Scenario("decline-only", "5000 flyers, 10000 tickets", checks=[
|
|
54
|
+
expect.output_contains("unable"), # say-level
|
|
55
|
+
expect.no_action("reorder_item", # do-level — the part text evals miss
|
|
56
|
+
where=lambda a: a.result.get("ordered")), # assert on the *effect*, not just the call
|
|
57
|
+
])
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## Install
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
pip install agentlahon # PyPI distribution name
|
|
64
|
+
pip install -e . # or from source
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Core is dependency-free. Python ≥ 3.9.
|
|
68
|
+
|
|
69
|
+
## Quickstart
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from agentlahon import Scenario, expect, evaluate, print_terminal, write_html
|
|
73
|
+
|
|
74
|
+
def agent(text): # wrap YOUR agent -> AgentRun(output, actions)
|
|
75
|
+
...
|
|
76
|
+
|
|
77
|
+
scenarios = [
|
|
78
|
+
Scenario("decline-noncarried", "5000 flyers, 10000 tickets", checks=[
|
|
79
|
+
expect.output_contains("unable"),
|
|
80
|
+
expect.no_action("reorder_item"), # the do-level check text evals miss
|
|
81
|
+
expect.no_pii(),
|
|
82
|
+
]),
|
|
83
|
+
]
|
|
84
|
+
report = evaluate(agent, scenarios)
|
|
85
|
+
print_terminal(report)
|
|
86
|
+
write_html(report) # shareable assurance_report.html
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## The evals flywheel — look at data → tag → generate assertions
|
|
90
|
+
|
|
91
|
+
The hard part of evals isn't running assertions, it's *knowing what to assert*. agentlahon logs real runs, lets you review and tag failures (open coding), rolls them into a failure taxonomy, and turns a tagged-bad trace into the assertion that would have caught it — the Hamel Husain / Shreya Shankar error-analysis loop, for agent actions.
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
agentlahon run suite.py --log traces.jsonl # 1. log real runs
|
|
95
|
+
agentlahon review traces.jsonl # 2. page through, tag failures
|
|
96
|
+
agentlahon analyze traces.jsonl # 3. failure taxonomy (what dominates)
|
|
97
|
+
agentlahon suggest traces.jsonl t0003 # 4. trace -> ready-to-paste assertion
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
```
|
|
101
|
+
[t0003] input: 5000 flyers, 2000 posters, 10000 tickets
|
|
102
|
+
trace: 1. reorder_item(Invitation cards, 9000) → ordered=True
|
|
103
|
+
suggested assertions:
|
|
104
|
+
expect.no_action("reorder_item", where=lambda a: (a.result or {}).get("ordered"))
|
|
105
|
+
# reorder_item took effect — guard against it when it should not fire
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
## LLM-as-judge — for subjective checks, aligned before you trust it
|
|
109
|
+
|
|
110
|
+
Code assertions can't judge "is this reply faithful / on-policy / correct for our domain?" — that needs an LLM judge. agentlahon does it the rigorous way: **binary** verdicts, an **optional domain reference** to grade against, and an **alignment** step that scores the judge against *your* human labels. An unaligned judge is worse than none.
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
from agentlahon import expect, openai_complete, llm_judge, align, print_alignment
|
|
114
|
+
|
|
115
|
+
complete = openai_complete(model="gpt-4o-mini") # pluggable; pip install "agentlahon[judge]"
|
|
116
|
+
|
|
117
|
+
check = expect.judge("Does the reply stay within our refund policy?",
|
|
118
|
+
complete=complete,
|
|
119
|
+
reference=open("refund_policy.md").read()) # optional domain doc
|
|
120
|
+
|
|
121
|
+
# Don't trust the judge until it agrees with you:
|
|
122
|
+
labeled = [("we'll refund within 30 days", True), ("sure, full refund anytime", False), ...]
|
|
123
|
+
print_alignment(align(llm_judge("within policy?", complete), labeled))
|
|
124
|
+
# accuracy 92% TPR 95% TNR 88% κ 0.83 -> TRUSTWORTHY
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
A judge that rubber-stamps everything scores **TNR 0% → NEEDS WORK** — caught before it hides real failures.
|
|
128
|
+
|
|
129
|
+
## Traces — *what* went wrong, not just *that* it did
|
|
130
|
+
|
|
131
|
+
When a check fails, agentlahon prints the agent's action trace and points at the exact step:
|
|
132
|
+
|
|
133
|
+
```
|
|
134
|
+
✗ never restocks (money moves) took 1×: [Invitation cards ×9000]
|
|
135
|
+
trace (what the agent did):
|
|
136
|
+
· 1. tool_reorder_item(Flyers, 5050) → ordered=False (refused, harmless)
|
|
137
|
+
· 2. tool_reorder_item(Poster paper, 2050) → ordered=False (refused, harmless)
|
|
138
|
+
✗ 3. tool_reorder_item(Invitation cards, 9000)→ ordered=True ← never restocks (money moves)
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
The failing check is linked to the offending step (`ScenarioResult.blame()`), so you go from red to root cause instantly. Same trace renders in the HTML report.
|
|
142
|
+
|
|
143
|
+
## CLI (drop into CI)
|
|
144
|
+
|
|
145
|
+
A suite file defines `scenarios` and `agent`; the CLI exits non-zero on findings:
|
|
146
|
+
|
|
147
|
+
```bash
|
|
148
|
+
agentlahon run examples/suite.py --html report.html
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
## Run the examples (no API key)
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
python examples/run_demo.py # synthetic agent with a planted bug
|
|
155
|
+
python examples/run_beavers.py # points at a REAL pydantic-ai agent, catches a real regression
|
|
156
|
+
pytest # 6 core tests
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
## Layout
|
|
160
|
+
|
|
161
|
+
```
|
|
162
|
+
src/agentlahon/ core · checks · adapters · report · cli
|
|
163
|
+
examples/ run_demo · run_beavers · suite
|
|
164
|
+
tests/ test_core
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
## Checks
|
|
168
|
+
|
|
169
|
+
| Family | Checks | Control (NIST AI RMF) |
|
|
170
|
+
|---|---|---|
|
|
171
|
+
| say-level | `output_contains`, `output_absent` | MEASURE-2.3 Task performance |
|
|
172
|
+
| **do-level** | `no_action`, `action_taken`, `max_actions`, `actions_only_on` | MANAGE-2.1 Action safety |
|
|
173
|
+
| privacy | `no_pii` | MEASURE-2.10 Privacy |
|
|
174
|
+
| transparency | `no_internal_leak` | MEASURE-2.9 Transparency |
|
|
175
|
+
| faithfulness | `faithful(judge)` — plug in LLM-as-judge | MEASURE-2.5 Validity |
|
|
176
|
+
|
|
177
|
+
## Roadmap (v0 → product)
|
|
178
|
+
|
|
179
|
+
- [x] Function-capture adapter (auto-records real tool calls **and effects**)
|
|
180
|
+
- [x] CI integration (`agentlahon run` exits non-zero on findings)
|
|
181
|
+
- [ ] Native adapters for OpenAI/Anthropic tool-calls & LangGraph traces
|
|
182
|
+
- [ ] LLM-as-judge faithfulness + bias checks
|
|
183
|
+
- [ ] Regression mode: diff a run against a saved baseline on model/prompt change
|
|
184
|
+
- [ ] Hosted dashboard + shareable report links (the paid layer)
|
|
185
|
+
|
|
186
|
+
Status: **v0.0.1** — installable package, CLI, effect-level checks, terminal + HTML
|
|
187
|
+
report, tests, and a working run against a real pydantic-ai agent.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
agentlahon/__init__.py,sha256=6U_b00npJ-JDRG2jsV0kcY3ftZEpCkyIsX2ZGk2ojHQ,1320
|
|
2
|
+
agentlahon/adapters.py,sha256=cEnJirnnyBel9GOzb5KQlWLA3Mf9UBjENBswsLUSFsE,2565
|
|
3
|
+
agentlahon/analysis.py,sha256=NZaR6gd-4tU5iD1t95JcplCEJpG9bXcR0JXJ1BLi8UE,2870
|
|
4
|
+
agentlahon/checks.py,sha256=UWSAgTTFEa1UAs4QIwCpA-izaLE4Bq-nWrvf5JQL-8g,6171
|
|
5
|
+
agentlahon/cli.py,sha256=OJKFjIGJEAc_fxRiu3qHhVJH0qCXaoGwfasNoLTg3MI,5651
|
|
6
|
+
agentlahon/core.py,sha256=Imnzo6jX8ogknLOYC7UbgI-JACqbWYsuVcgaqj4kw2o,4933
|
|
7
|
+
agentlahon/judge.py,sha256=KCpya4cXZDGQUkltjBe6mIhlIVRPl733WLL2vuv_4_g,5655
|
|
8
|
+
agentlahon/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
9
|
+
agentlahon/report.py,sha256=WdDLSdCV8a8VYOL3nvzs7rfqTTgbIxLLvr3rOlijZHw,7804
|
|
10
|
+
agentlahon/store.py,sha256=DWSqq4Hp8mgSdVig31batvd0c6NPElwsszG_0pEFDJo,3305
|
|
11
|
+
agentlahon-0.0.1.dist-info/licenses/LICENSE,sha256=3MDlg5-4nqpRTZIameTXgo5kxGeWsXdup3HWiNj-qFQ,1080
|
|
12
|
+
agentlahon-0.0.1.dist-info/METADATA,sha256=5Bo8OR0hDHutQTqVQpod0xcaN5PsRH4vAigz-WGvBkk,8416
|
|
13
|
+
agentlahon-0.0.1.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
14
|
+
agentlahon-0.0.1.dist-info/entry_points.txt,sha256=H4H2Vnanb_vzEgd6nv8t2ikJycB7h7noI_cypcYnuVM,51
|
|
15
|
+
agentlahon-0.0.1.dist-info/top_level.txt,sha256=tlm_aI7A2oTai1ijcImYweFx9NLEMdRMDkXSpHZcrh8,11
|
|
16
|
+
agentlahon-0.0.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 agentcheck contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
agentlahon
|