hieevas 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hieevas-0.1.0/LICENSE +21 -0
- hieevas-0.1.0/PKG-INFO +262 -0
- hieevas-0.1.0/README.md +202 -0
- hieevas-0.1.0/pyproject.toml +66 -0
- hieevas-0.1.0/setup.cfg +4 -0
- hieevas-0.1.0/src/hieevas/__init__.py +34 -0
- hieevas-0.1.0/src/hieevas/adapters/__init__.py +2 -0
- hieevas-0.1.0/src/hieevas/adapters/langchain.py +103 -0
- hieevas-0.1.0/src/hieevas/adapters/otel.py +503 -0
- hieevas-0.1.0/src/hieevas/dashboard.py +323 -0
- hieevas-0.1.0/src/hieevas/evaluate.py +76 -0
- hieevas-0.1.0/src/hieevas/governance.py +127 -0
- hieevas-0.1.0/src/hieevas/live.py +322 -0
- hieevas-0.1.0/src/hieevas/metrics/__init__.py +23 -0
- hieevas-0.1.0/src/hieevas/metrics/base.py +83 -0
- hieevas-0.1.0/src/hieevas/metrics/definitions.py +375 -0
- hieevas-0.1.0/src/hieevas/metrics/guide.py +232 -0
- hieevas-0.1.0/src/hieevas/py.typed +0 -0
- hieevas-0.1.0/src/hieevas/recorder.py +175 -0
- hieevas-0.1.0/src/hieevas/report.py +120 -0
- hieevas-0.1.0/src/hieevas/scoring.py +144 -0
- hieevas-0.1.0/src/hieevas/serve.py +159 -0
- hieevas-0.1.0/src/hieevas/stress.py +77 -0
- hieevas-0.1.0/src/hieevas/trace.py +136 -0
- hieevas-0.1.0/src/hieevas.egg-info/PKG-INFO +262 -0
- hieevas-0.1.0/src/hieevas.egg-info/SOURCES.txt +35 -0
- hieevas-0.1.0/src/hieevas.egg-info/dependency_links.txt +1 -0
- hieevas-0.1.0/src/hieevas.egg-info/entry_points.txt +2 -0
- hieevas-0.1.0/src/hieevas.egg-info/requires.txt +38 -0
- hieevas-0.1.0/src/hieevas.egg-info/top_level.txt +1 -0
- hieevas-0.1.0/tests/test_langchain_adapter.py +62 -0
- hieevas-0.1.0/tests/test_live.py +176 -0
- hieevas-0.1.0/tests/test_metrics.py +127 -0
- hieevas-0.1.0/tests/test_otel_adapter.py +94 -0
- hieevas-0.1.0/tests/test_report_and_io.py +53 -0
- hieevas-0.1.0/tests/test_scoring.py +55 -0
- hieevas-0.1.0/tests/test_stress.py +97 -0
hieevas-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Sampathkumar T.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
hieevas-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hieevas
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Hierarchical evaluation of agentic systems (HIEEVAS): 27 process, cost, robustness, safety and governance metrics at agent, interaction and system level, computed from execution traces of LangGraph and OpenTelemetry-instrumented agents.
|
|
5
|
+
Author-email: "Sampathkumar T." <heurius.tech@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/heurius/hieevas
|
|
8
|
+
Project-URL: Repository, https://github.com/heurius/hieevas
|
|
9
|
+
Project-URL: Issues, https://github.com/heurius/hieevas/issues
|
|
10
|
+
Project-URL: Documentation, https://github.com/heurius/hieevas/blob/main/METRICS.md
|
|
11
|
+
Project-URL: Changelog, https://github.com/heurius/hieevas/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: agentic-ai,ai-agents,llm,evaluation,metrics,langgraph,langchain,rag,opentelemetry,observability,ai-governance,prompt-injection,vertex-ai
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Software Development :: Testing
|
|
25
|
+
Classifier: Topic :: System :: Monitoring
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: >=3.10
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
License-File: LICENSE
|
|
30
|
+
Provides-Extra: langchain
|
|
31
|
+
Requires-Dist: langchain-core>=0.3; extra == "langchain"
|
|
32
|
+
Provides-Extra: langgraph
|
|
33
|
+
Requires-Dist: langchain-core>=0.3; extra == "langgraph"
|
|
34
|
+
Requires-Dist: langgraph>=0.2; extra == "langgraph"
|
|
35
|
+
Provides-Extra: pandas
|
|
36
|
+
Requires-Dist: pandas>=2.0; extra == "pandas"
|
|
37
|
+
Provides-Extra: otel
|
|
38
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "otel"
|
|
39
|
+
Provides-Extra: agentcore
|
|
40
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "agentcore"
|
|
41
|
+
Requires-Dist: boto3>=1.34; extra == "agentcore"
|
|
42
|
+
Provides-Extra: vertex
|
|
43
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "vertex"
|
|
44
|
+
Requires-Dist: google-cloud-trace>=1.13; extra == "vertex"
|
|
45
|
+
Provides-Extra: live
|
|
46
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "live"
|
|
47
|
+
Requires-Dist: opentelemetry-exporter-otlp-proto-http>=1.20; extra == "live"
|
|
48
|
+
Requires-Dist: openinference-instrumentation-langchain>=0.1.20; extra == "live"
|
|
49
|
+
Provides-Extra: gcp
|
|
50
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "gcp"
|
|
51
|
+
Requires-Dist: opentelemetry-exporter-gcp-trace>=1.6; extra == "gcp"
|
|
52
|
+
Requires-Dist: google-cloud-trace>=1.13; extra == "gcp"
|
|
53
|
+
Requires-Dist: openinference-instrumentation-langchain>=0.1.20; extra == "gcp"
|
|
54
|
+
Provides-Extra: dev
|
|
55
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
56
|
+
Requires-Dist: langchain-core>=0.3; extra == "dev"
|
|
57
|
+
Requires-Dist: langgraph>=0.2; extra == "dev"
|
|
58
|
+
Requires-Dist: opentelemetry-sdk>=1.20; extra == "dev"
|
|
59
|
+
Dynamic: license-file
|
|
60
|
+
|
|
61
|
+
# hieevas
|
|
62
|
+
|
|
63
|
+
**HIE**rarchical **EV**aluation of **A**gentic **S**ystems: evaluate **how** an agentic AI or RAG
|
|
64
|
+
pipeline reaches its answers, not only **whether** it does.
|
|
65
|
+
|
|
66
|
+
Answer-quality tools such as Ragas judge the final answer and its grounding; hieevas measures the
|
|
67
|
+
path to it (cost, robustness under tool faults, prompt-injection resistance, governance and audit)
|
|
68
|
+
at three levels of the hierarchy: the single agent, the interaction between agents, and the whole
|
|
69
|
+
system. The two complement each other.
|
|
70
|
+
|
|
71
|
+
`hieevas` records what your pipeline does (LLM calls, tool calls, messages between agents,
|
|
72
|
+
reviews, approvals) and scores it on **27 metrics in 7 dimensions** at **agent, interaction and
|
|
73
|
+
system level**, following the multi-level evaluation framework in *Measuring the Process, Not Just
|
|
74
|
+
the Outcome* (Table 8). It works offline, has no required dependencies, and produces a
|
|
75
|
+
self-contained HTML dashboard.
|
|
76
|
+
|
|
77
|
+
| Dimension | Metrics |
|
|
78
|
+
|---|---|
|
|
79
|
+
| Effectiveness | M1 task success · M2 answer F1 · M3 consistency |
|
|
80
|
+
| Efficiency | M4 completion time · M5 tokens/task · M6 LLM calls · M7 tokens per success · M8 peak context |
|
|
81
|
+
| Planning & reasoning | M9 steps · M10 redundant calls · M11 plan adherence · M12 tool-call validity |
|
|
82
|
+
| Robustness & recovery | M13 fault-induced drop · M14 recovery rate · M15 retries per error |
|
|
83
|
+
| Coordination | M16 hand-over success · M17 reviewer rejections · M18 correction success · M19 communication overhead |
|
|
84
|
+
| Safety & security | M20 refusal rate · M21 injection success · M22 unauthorised high-risk attempts |
|
|
85
|
+
| Transparency & governance | M23 rationale coverage · M24 rationale quality · M25 audit completeness · M26 approval triggers · M27 budget violations |
|
|
86
|
+
|
|
87
|
+
When the data for a metric was not recorded, the metric returns *not available* with the reason,
|
|
88
|
+
never a misleading zero. [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md) explains every metric in plain language: what it
|
|
89
|
+
measures, how it is computed, and which part of a LangGraph run it reads (input, LLM call, tool call,
|
|
90
|
+
final state). The same text appears under *What is this?* on each dashboard card.
|
|
91
|
+
|
|
92
|
+
## Install
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
pip install hieevas # core, no dependencies
|
|
96
|
+
pip install "hieevas[langgraph]" # + LangChain / LangGraph callback adapter
|
|
97
|
+
pip install "hieevas[live]" # + live tracing of LangGraph apps (local mode, any OTLP collector)
|
|
98
|
+
pip install "hieevas[gcp]" # + Cloud Run / VM -> Cloud Trace, and reading Cloud Trace (incl. Agent Engine)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Until the first PyPI release, install from GitHub:
|
|
102
|
+
`pip install "hieevas[live] @ git+https://github.com/heurius/hieevas"`.
|
|
103
|
+
For development: `git clone https://github.com/heurius/hieevas && pip install -e "./hieevas[dev]"`.
|
|
104
|
+
|
|
105
|
+
## 0. Live dashboards for any LangGraph app
|
|
106
|
+
|
|
107
|
+
Add one line at start-up; tag requests only if you want known-answer or stress probes.
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
import hieevas
|
|
111
|
+
|
|
112
|
+
session = hieevas.init("local") # or "cloud" (Cloud Run / VM) or "vertex" (Agent Engine)
|
|
113
|
+
graph.invoke(inputs) # live traffic: no changes
|
|
114
|
+
graph.invoke(inputs, config=hieevas.run_config(task_id="q1", condition="C1", reference="Paris")) # probe
|
|
115
|
+
session.save_dashboard("dashboard.html")
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
python -m hieevas.serve --source file:traces/spans.jsonl # local mode
|
|
120
|
+
python -m hieevas.serve --source cloud-trace:MY_PROJECT --tool-risk send_email # cloud and vertex modes
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
| Mode | App and LLM | Traces written by | Dashboard reads |
|
|
124
|
+
|---|---|---|---|
|
|
125
|
+
| `local` | both on this machine (e.g. Ollama) | `hieevas.init("local")` → JSON-lines file | `file:traces/spans.jsonl` |
|
|
126
|
+
| `cloud` | app on Cloud Run / a VM, LLM via an API | `hieevas.init("cloud")` → Cloud Trace (or `endpoint=` any OTLP collector) | `cloud-trace:PROJECT` |
|
|
127
|
+
| `vertex` | agent on Vertex AI Agent Engine | `LanggraphAgent(..., enable_tracing=True)` → Cloud Trace | `cloud-trace:PROJECT` |
|
|
128
|
+
|
|
129
|
+
All three produce the same OpenInference spans, so one reader and one dashboard serve them.
|
|
130
|
+
Tags passed with `run_config` travel in the LangGraph config metadata and are read back
|
|
131
|
+
from the spans, which also works on Agent Engine where the app does not own its spans.
|
|
132
|
+
The dashboard server (`/`, `/report.json`, `/healthz`; query `?hours=6&group_by=architecture,hour`)
|
|
133
|
+
needs only the standard library; on Cloud Run deploy it with `--no-allow-unauthenticated`.
|
|
134
|
+
|
|
135
|
+
**What live traffic can and cannot show.** Real user requests have no reference answers, so
|
|
136
|
+
they yield the efficiency, planning, coordination, high-risk-attempt and governance metrics.
|
|
137
|
+
Task success, recovery, refusal and injection metrics come from probes: requests tagged with a
|
|
138
|
+
condition (`C1` known answer, `C2` tool faults, `C3` injection, `C4` out-of-policy) sent with
|
|
139
|
+
`hieevas.send_probes(...)`, for example on a schedule. For `C4` runs without an explicit
|
|
140
|
+
flag, refusal is scored as *declined in words and no high-risk tool attempted*.
|
|
141
|
+
|
|
142
|
+
Examples for each mode are in [`examples/`](https://github.com/heurius/hieevas/tree/main/examples): `local_ollama.py`, `cloud_run/`,
|
|
143
|
+
`vertex_agent_engine.py`, and a stand-alone dashboard container in `dashboard/`.
|
|
144
|
+
|
|
145
|
+
## Using hieevas in production
|
|
146
|
+
|
|
147
|
+
hieevas is alpha software, provided "as is" under the MIT licence without warranty. Before
|
|
148
|
+
running it next to real users, check these four points:
|
|
149
|
+
|
|
150
|
+
1. **The approval gate blocks real actions by default.** `stress_tools(..., high_risk=[...])`
|
|
151
|
+
routes every call to a high-risk tool through `approval`, for probes *and* real users, and the
|
|
152
|
+
default `deny_all` refuses them all. In production pass your own approval function (for example
|
|
153
|
+
one that asks a person), or `approval=approve_all` from `hieevas.governance` to record without
|
|
154
|
+
blocking.
|
|
155
|
+
2. **Stress conditions are for probes only.** Tool faults (`C2`) and injected text (`C3`) are
|
|
156
|
+
applied only to requests tagged with that condition. Never let end users set tags: the Cloud
|
|
157
|
+
Run example rejects tags unless `HIEEVAS_ACCEPT_TAGS=1`, and probes should come from a
|
|
158
|
+
trusted caller.
|
|
159
|
+
3. **Traces contain user data.** Spans hold prompts, retrieved passages, tool arguments and
|
|
160
|
+
answers, which may include personal or confidential information. Treat trace files and Cloud
|
|
161
|
+
Trace like application logs: restrict access, set retention, sample (`sample_rate=`) and
|
|
162
|
+
follow your privacy obligations. The dashboard server has no login of its own; locally it
|
|
163
|
+
listens on 127.0.0.1 only, and on Cloud Run it must be deployed with
|
|
164
|
+
`--no-allow-unauthenticated` or behind IAP.
|
|
165
|
+
4. **Existing OpenTelemetry setups are joined, not replaced.** If the application already
|
|
166
|
+
configured a tracer provider, `hieevas.init()` adds its exporter to that provider and warns;
|
|
167
|
+
the application's service name and sampling stay in force.
|
|
168
|
+
|
|
169
|
+
Metric values are measurements from recorded behaviour, not certifications: keyword-based
|
|
170
|
+
refusal detection and phrase-matching of answers are heuristics (see [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md)).
|
|
171
|
+
See [SECURITY.md](https://github.com/heurius/hieevas/blob/main/SECURITY.md) to report a vulnerability.
|
|
172
|
+
|
|
173
|
+
## 1. Any pipeline (framework-agnostic)
|
|
174
|
+
|
|
175
|
+
```python
|
|
176
|
+
from hieevas import Recorder, evaluate
|
|
177
|
+
|
|
178
|
+
rec = Recorder(app="my-rag", architecture="A1", model="qwen2.5:1.5b")
|
|
179
|
+
with rec.run("q1", reference="Paris") as run:
|
|
180
|
+
run.llm_call(agent="retriever", input_tokens=120, output_tokens=18, latency=0.9)
|
|
181
|
+
run.tool_call("search_docs", {"query": "capital of France"}, rationale="need a source")
|
|
182
|
+
run.answer("Paris")
|
|
183
|
+
|
|
184
|
+
report = evaluate(rec.runs, group_by=("architecture", "condition"))
|
|
185
|
+
print(report.summary())
|
|
186
|
+
report.to_html("dashboard.html") # offline dashboard
|
|
187
|
+
report.to_csv("metrics.csv")
|
|
188
|
+
rec.save("traces.jsonl"); rec.save_csv("logs/") # runs.csv + steps.csv
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
Multi-agent pipelines can also record `run.plan([...])`, `run.plan_step_done(i)`,
|
|
192
|
+
`run.message(sender, receiver, text)`, `run.review("accept"|"reject")` and
|
|
193
|
+
`run.mark(refused=..., injection_followed=...)`.
|
|
194
|
+
|
|
195
|
+
## 2. LangChain / LangGraph
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
from hieevas.adapters.langchain import HieevasCallbackHandler
|
|
199
|
+
|
|
200
|
+
with rec.run("q1", reference="Paris") as run:
|
|
201
|
+
out = graph.invoke(inputs, config={"callbacks": [HieevasCallbackHandler(run)],
|
|
202
|
+
"recursion_limit": 40}) # recursion limit = step budget (M27)
|
|
203
|
+
run.answer(out["messages"][-1].content)
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
The handler records every LLM call (tokens, latency, agent = LangGraph node) and every tool call
|
|
207
|
+
(status, arguments, and the text the model wrote before calling it, used as the rationale).
|
|
208
|
+
|
|
209
|
+
## 3. Governance layer and test conditions
|
|
210
|
+
|
|
211
|
+
```python
|
|
212
|
+
from hieevas import governed_tool, inject, injection_followed
|
|
213
|
+
|
|
214
|
+
search = governed_tool(search_fn, name="search_docs", fault_rate=0.3) # C2 tool faults
|
|
215
|
+
send_email = governed_tool(send_fn, name="send_email", risk="high") # approval gate (denies by default)
|
|
216
|
+
docs = inject(retrieved_docs) # C3 prompt injection
|
|
217
|
+
run.mark(injection_followed=injection_followed(run.record))
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
## 4. OpenTelemetry: AgentCore, Vertex AI and any OTel backend
|
|
221
|
+
|
|
222
|
+
One trace becomes one run. Spans using the OpenTelemetry GenAI conventions (`gen_ai.*`),
|
|
223
|
+
OpenInference or OpenLLMetry are understood.
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
# a) In-process, wherever the agent runs (locally, AgentCore Runtime, Vertex Agent Engine)
|
|
227
|
+
from hieevas.adapters.otel import HieevasSpanProcessor
|
|
228
|
+
proc = HieevasSpanProcessor()
|
|
229
|
+
tracer_provider.add_span_processor(proc) # alongside your normal exporter
|
|
230
|
+
... # run the agent
|
|
231
|
+
report = evaluate(proc.runs(architecture="A1"))
|
|
232
|
+
|
|
233
|
+
# b) From files written by an OpenTelemetry Collector (file exporter)
|
|
234
|
+
from hieevas.adapters.otel import load_otlp_json, spans_to_runs
|
|
235
|
+
runs = spans_to_runs(load_otlp_json("traces.json"))
|
|
236
|
+
|
|
237
|
+
# c) Amazon Bedrock AgentCore: spans in CloudWatch (log group aws/spans) pip install "hieevas[agentcore]"
|
|
238
|
+
from hieevas.adapters.otel import fetch_cloudwatch_spans
|
|
239
|
+
runs = spans_to_runs(fetch_cloudwatch_spans(start_ms, end_ms, region="us-east-1"))
|
|
240
|
+
|
|
241
|
+
# d) Google Vertex AI Agent Engine: spans in Cloud Trace pip install "hieevas[vertex]"
|
|
242
|
+
from hieevas.adapters.otel import fetch_cloud_trace_spans
|
|
243
|
+
runs = spans_to_runs(fetch_cloud_trace_spans("my-gcp-project"))
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
Set `hieevas.task_id`, `hieevas.condition`, `hieevas.reference` and `hieevas.architecture`
|
|
247
|
+
as span attributes to control grouping and scoring; `hieevas.rationale` and `hieevas.risk`
|
|
248
|
+
on tool spans enable M23 and M22. The CloudWatch and Cloud Trace loaders are experimental:
|
|
249
|
+
check the span record format for your platform version.
|
|
250
|
+
|
|
251
|
+
## Example
|
|
252
|
+
|
|
253
|
+
`../rag_eval/` runs a RAG pipeline over four PDF papers with a local Ollama model in three
|
|
254
|
+
architectures (single agent, planner–executor, planner–executor–reviewer, built with LangGraph)
|
|
255
|
+
under four conditions, and writes the dashboard.
|
|
256
|
+
|
|
257
|
+
## Notes
|
|
258
|
+
|
|
259
|
+
- M20 and M24 are rubric-scored. `detect_refusal` is a keyword heuristic for screening only;
|
|
260
|
+
confirm codes with human raters and report agreement with `cohen_kappa`.
|
|
261
|
+
- Token counts come from the model provider; message tokens are estimated (4 characters ≈ 1 token)
|
|
262
|
+
when not supplied.
|
hieevas-0.1.0/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
# hieevas
|
|
2
|
+
|
|
3
|
+
**HIE**rarchical **EV**aluation of **A**gentic **S**ystems: evaluate **how** an agentic AI or RAG
|
|
4
|
+
pipeline reaches its answers, not only **whether** it does.
|
|
5
|
+
|
|
6
|
+
Answer-quality tools such as Ragas judge the final answer and its grounding; hieevas measures the
|
|
7
|
+
path to it (cost, robustness under tool faults, prompt-injection resistance, governance and audit)
|
|
8
|
+
at three levels of the hierarchy: the single agent, the interaction between agents, and the whole
|
|
9
|
+
system. The two complement each other.
|
|
10
|
+
|
|
11
|
+
`hieevas` records what your pipeline does (LLM calls, tool calls, messages between agents,
|
|
12
|
+
reviews, approvals) and scores it on **27 metrics in 7 dimensions** at **agent, interaction and
|
|
13
|
+
system level**, following the multi-level evaluation framework in *Measuring the Process, Not Just
|
|
14
|
+
the Outcome* (Table 8). It works offline, has no required dependencies, and produces a
|
|
15
|
+
self-contained HTML dashboard.
|
|
16
|
+
|
|
17
|
+
| Dimension | Metrics |
|
|
18
|
+
|---|---|
|
|
19
|
+
| Effectiveness | M1 task success · M2 answer F1 · M3 consistency |
|
|
20
|
+
| Efficiency | M4 completion time · M5 tokens/task · M6 LLM calls · M7 tokens per success · M8 peak context |
|
|
21
|
+
| Planning & reasoning | M9 steps · M10 redundant calls · M11 plan adherence · M12 tool-call validity |
|
|
22
|
+
| Robustness & recovery | M13 fault-induced drop · M14 recovery rate · M15 retries per error |
|
|
23
|
+
| Coordination | M16 hand-over success · M17 reviewer rejections · M18 correction success · M19 communication overhead |
|
|
24
|
+
| Safety & security | M20 refusal rate · M21 injection success · M22 unauthorised high-risk attempts |
|
|
25
|
+
| Transparency & governance | M23 rationale coverage · M24 rationale quality · M25 audit completeness · M26 approval triggers · M27 budget violations |
|
|
26
|
+
|
|
27
|
+
When the data for a metric was not recorded, the metric returns *not available* with the reason,
|
|
28
|
+
never a misleading zero. [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md) explains every metric in plain language: what it
|
|
29
|
+
measures, how it is computed, and which part of a LangGraph run it reads (input, LLM call, tool call,
|
|
30
|
+
final state). The same text appears under *What is this?* on each dashboard card.
|
|
31
|
+
|
|
32
|
+
## Install
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install hieevas # core, no dependencies
|
|
36
|
+
pip install "hieevas[langgraph]" # + LangChain / LangGraph callback adapter
|
|
37
|
+
pip install "hieevas[live]" # + live tracing of LangGraph apps (local mode, any OTLP collector)
|
|
38
|
+
pip install "hieevas[gcp]" # + Cloud Run / VM -> Cloud Trace, and reading Cloud Trace (incl. Agent Engine)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Until the first PyPI release, install from GitHub:
|
|
42
|
+
`pip install "hieevas[live] @ git+https://github.com/heurius/hieevas"`.
|
|
43
|
+
For development: `git clone https://github.com/heurius/hieevas && pip install -e "./hieevas[dev]"`.
|
|
44
|
+
|
|
45
|
+
## 0. Live dashboards for any LangGraph app
|
|
46
|
+
|
|
47
|
+
Add one line at start-up; tag requests only if you want known-answer or stress probes.
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
import hieevas
|
|
51
|
+
|
|
52
|
+
session = hieevas.init("local") # or "cloud" (Cloud Run / VM) or "vertex" (Agent Engine)
|
|
53
|
+
graph.invoke(inputs) # live traffic: no changes
|
|
54
|
+
graph.invoke(inputs, config=hieevas.run_config(task_id="q1", condition="C1", reference="Paris")) # probe
|
|
55
|
+
session.save_dashboard("dashboard.html")
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
python -m hieevas.serve --source file:traces/spans.jsonl # local mode
|
|
60
|
+
python -m hieevas.serve --source cloud-trace:MY_PROJECT --tool-risk send_email # cloud and vertex modes
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
| Mode | App and LLM | Traces written by | Dashboard reads |
|
|
64
|
+
|---|---|---|---|
|
|
65
|
+
| `local` | both on this machine (e.g. Ollama) | `hieevas.init("local")` → JSON-lines file | `file:traces/spans.jsonl` |
|
|
66
|
+
| `cloud` | app on Cloud Run / a VM, LLM via an API | `hieevas.init("cloud")` → Cloud Trace (or `endpoint=` any OTLP collector) | `cloud-trace:PROJECT` |
|
|
67
|
+
| `vertex` | agent on Vertex AI Agent Engine | `LanggraphAgent(..., enable_tracing=True)` → Cloud Trace | `cloud-trace:PROJECT` |
|
|
68
|
+
|
|
69
|
+
All three produce the same OpenInference spans, so one reader and one dashboard serve them.
|
|
70
|
+
Tags passed with `run_config` travel in the LangGraph config metadata and are read back
|
|
71
|
+
from the spans, which also works on Agent Engine where the app does not own its spans.
|
|
72
|
+
The dashboard server (`/`, `/report.json`, `/healthz`; query `?hours=6&group_by=architecture,hour`)
|
|
73
|
+
needs only the standard library; on Cloud Run deploy it with `--no-allow-unauthenticated`.
|
|
74
|
+
|
|
75
|
+
**What live traffic can and cannot show.** Real user requests have no reference answers, so
|
|
76
|
+
they yield the efficiency, planning, coordination, high-risk-attempt and governance metrics.
|
|
77
|
+
Task success, recovery, refusal and injection metrics come from probes: requests tagged with a
|
|
78
|
+
condition (`C1` known answer, `C2` tool faults, `C3` injection, `C4` out-of-policy) sent with
|
|
79
|
+
`hieevas.send_probes(...)`, for example on a schedule. For `C4` runs without an explicit
|
|
80
|
+
flag, refusal is scored as *declined in words and no high-risk tool attempted*.
|
|
81
|
+
|
|
82
|
+
Examples for each mode are in [`examples/`](https://github.com/heurius/hieevas/tree/main/examples): `local_ollama.py`, `cloud_run/`,
|
|
83
|
+
`vertex_agent_engine.py`, and a stand-alone dashboard container in `dashboard/`.
|
|
84
|
+
|
|
85
|
+
## Using hieevas in production
|
|
86
|
+
|
|
87
|
+
hieevas is alpha software, provided "as is" under the MIT licence without warranty. Before
|
|
88
|
+
running it next to real users, check these four points:
|
|
89
|
+
|
|
90
|
+
1. **The approval gate blocks real actions by default.** `stress_tools(..., high_risk=[...])`
|
|
91
|
+
routes every call to a high-risk tool through `approval`, for probes *and* real users, and the
|
|
92
|
+
default `deny_all` refuses them all. In production pass your own approval function (for example
|
|
93
|
+
one that asks a person), or `approval=approve_all` from `hieevas.governance` to record without
|
|
94
|
+
blocking.
|
|
95
|
+
2. **Stress conditions are for probes only.** Tool faults (`C2`) and injected text (`C3`) are
|
|
96
|
+
applied only to requests tagged with that condition. Never let end users set tags: the Cloud
|
|
97
|
+
Run example rejects tags unless `HIEEVAS_ACCEPT_TAGS=1`, and probes should come from a
|
|
98
|
+
trusted caller.
|
|
99
|
+
3. **Traces contain user data.** Spans hold prompts, retrieved passages, tool arguments and
|
|
100
|
+
answers, which may include personal or confidential information. Treat trace files and Cloud
|
|
101
|
+
Trace like application logs: restrict access, set retention, sample (`sample_rate=`) and
|
|
102
|
+
follow your privacy obligations. The dashboard server has no login of its own; locally it
|
|
103
|
+
listens on 127.0.0.1 only, and on Cloud Run it must be deployed with
|
|
104
|
+
`--no-allow-unauthenticated` or behind IAP.
|
|
105
|
+
4. **Existing OpenTelemetry setups are joined, not replaced.** If the application already
|
|
106
|
+
configured a tracer provider, `hieevas.init()` adds its exporter to that provider and warns;
|
|
107
|
+
the application's service name and sampling stay in force.
|
|
108
|
+
|
|
109
|
+
Metric values are measurements from recorded behaviour, not certifications: keyword-based
|
|
110
|
+
refusal detection and phrase-matching of answers are heuristics (see [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md)).
|
|
111
|
+
See [SECURITY.md](https://github.com/heurius/hieevas/blob/main/SECURITY.md) to report a vulnerability.
|
|
112
|
+
|
|
113
|
+
## 1. Any pipeline (framework-agnostic)
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from hieevas import Recorder, evaluate
|
|
117
|
+
|
|
118
|
+
rec = Recorder(app="my-rag", architecture="A1", model="qwen2.5:1.5b")
|
|
119
|
+
with rec.run("q1", reference="Paris") as run:
|
|
120
|
+
run.llm_call(agent="retriever", input_tokens=120, output_tokens=18, latency=0.9)
|
|
121
|
+
run.tool_call("search_docs", {"query": "capital of France"}, rationale="need a source")
|
|
122
|
+
run.answer("Paris")
|
|
123
|
+
|
|
124
|
+
report = evaluate(rec.runs, group_by=("architecture", "condition"))
|
|
125
|
+
print(report.summary())
|
|
126
|
+
report.to_html("dashboard.html") # offline dashboard
|
|
127
|
+
report.to_csv("metrics.csv")
|
|
128
|
+
rec.save("traces.jsonl"); rec.save_csv("logs/") # runs.csv + steps.csv
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Multi-agent pipelines can also record `run.plan([...])`, `run.plan_step_done(i)`,
|
|
132
|
+
`run.message(sender, receiver, text)`, `run.review("accept"|"reject")` and
|
|
133
|
+
`run.mark(refused=..., injection_followed=...)`.
|
|
134
|
+
|
|
135
|
+
## 2. LangChain / LangGraph
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
from hieevas.adapters.langchain import HieevasCallbackHandler
|
|
139
|
+
|
|
140
|
+
with rec.run("q1", reference="Paris") as run:
|
|
141
|
+
out = graph.invoke(inputs, config={"callbacks": [HieevasCallbackHandler(run)],
|
|
142
|
+
"recursion_limit": 40}) # recursion limit = step budget (M27)
|
|
143
|
+
run.answer(out["messages"][-1].content)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
The handler records every LLM call (tokens, latency, agent = LangGraph node) and every tool call
|
|
147
|
+
(status, arguments, and the text the model wrote before calling it, used as the rationale).
|
|
148
|
+
|
|
149
|
+
## 3. Governance layer and test conditions
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
from hieevas import governed_tool, inject, injection_followed
|
|
153
|
+
|
|
154
|
+
search = governed_tool(search_fn, name="search_docs", fault_rate=0.3) # C2 tool faults
|
|
155
|
+
send_email = governed_tool(send_fn, name="send_email", risk="high") # approval gate (denies by default)
|
|
156
|
+
docs = inject(retrieved_docs) # C3 prompt injection
|
|
157
|
+
run.mark(injection_followed=injection_followed(run.record))
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
## 4. OpenTelemetry: AgentCore, Vertex AI and any OTel backend
|
|
161
|
+
|
|
162
|
+
One trace becomes one run. Spans using the OpenTelemetry GenAI conventions (`gen_ai.*`),
|
|
163
|
+
OpenInference or OpenLLMetry are understood.
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
# a) In-process, wherever the agent runs (locally, AgentCore Runtime, Vertex Agent Engine)
|
|
167
|
+
from hieevas.adapters.otel import HieevasSpanProcessor
|
|
168
|
+
proc = HieevasSpanProcessor()
|
|
169
|
+
tracer_provider.add_span_processor(proc) # alongside your normal exporter
|
|
170
|
+
... # run the agent
|
|
171
|
+
report = evaluate(proc.runs(architecture="A1"))
|
|
172
|
+
|
|
173
|
+
# b) From files written by an OpenTelemetry Collector (file exporter)
|
|
174
|
+
from hieevas.adapters.otel import load_otlp_json, spans_to_runs
|
|
175
|
+
runs = spans_to_runs(load_otlp_json("traces.json"))
|
|
176
|
+
|
|
177
|
+
# c) Amazon Bedrock AgentCore: spans in CloudWatch (log group aws/spans) pip install "hieevas[agentcore]"
|
|
178
|
+
from hieevas.adapters.otel import fetch_cloudwatch_spans
|
|
179
|
+
runs = spans_to_runs(fetch_cloudwatch_spans(start_ms, end_ms, region="us-east-1"))
|
|
180
|
+
|
|
181
|
+
# d) Google Vertex AI Agent Engine: spans in Cloud Trace pip install "hieevas[vertex]"
|
|
182
|
+
from hieevas.adapters.otel import fetch_cloud_trace_spans
|
|
183
|
+
runs = spans_to_runs(fetch_cloud_trace_spans("my-gcp-project"))
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
Set `hieevas.task_id`, `hieevas.condition`, `hieevas.reference` and `hieevas.architecture`
|
|
187
|
+
as span attributes to control grouping and scoring; `hieevas.rationale` and `hieevas.risk`
|
|
188
|
+
on tool spans enable M23 and M22. The CloudWatch and Cloud Trace loaders are experimental:
|
|
189
|
+
check the span record format for your platform version.
|
|
190
|
+
|
|
191
|
+
## Example
|
|
192
|
+
|
|
193
|
+
`../rag_eval/` runs a RAG pipeline over four PDF papers with a local Ollama model in three
|
|
194
|
+
architectures (single agent, planner–executor, planner–executor–reviewer, built with LangGraph)
|
|
195
|
+
under four conditions, and writes the dashboard.
|
|
196
|
+
|
|
197
|
+
## Notes
|
|
198
|
+
|
|
199
|
+
- M20 and M24 are rubric-scored. `detect_refusal` is a keyword heuristic for screening only;
|
|
200
|
+
confirm codes with human raters and report agreement with `cohen_kappa`.
|
|
201
|
+
- Token counts come from the model provider; message tokens are estimated (4 characters ≈ 1 token)
|
|
202
|
+
when not supplied.
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hieevas"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Hierarchical evaluation of agentic systems (HIEEVAS): 27 process, cost, robustness, safety and governance metrics at agent, interaction and system level, computed from execution traces of LangGraph and OpenTelemetry-instrumented agents."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
authors = [{ name = "Sampathkumar T.", email = "heurius.tech@gmail.com" }]
|
|
14
|
+
keywords = ["agentic-ai", "ai-agents", "llm", "evaluation", "metrics", "langgraph", "langchain", "rag",
|
|
15
|
+
"opentelemetry", "observability", "ai-governance", "prompt-injection", "vertex-ai"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 3 - Alpha",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"Operating System :: OS Independent",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
23
|
+
"Programming Language :: Python :: 3.10",
|
|
24
|
+
"Programming Language :: Python :: 3.11",
|
|
25
|
+
"Programming Language :: Python :: 3.12",
|
|
26
|
+
"Programming Language :: Python :: 3.13",
|
|
27
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
28
|
+
"Topic :: Software Development :: Testing",
|
|
29
|
+
"Topic :: System :: Monitoring",
|
|
30
|
+
"Typing :: Typed",
|
|
31
|
+
]
|
|
32
|
+
dependencies = []
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
langchain = ["langchain-core>=0.3"]
|
|
36
|
+
langgraph = ["langchain-core>=0.3", "langgraph>=0.2"]
|
|
37
|
+
pandas = ["pandas>=2.0"]
|
|
38
|
+
otel = ["opentelemetry-sdk>=1.20"]
|
|
39
|
+
agentcore = ["opentelemetry-sdk>=1.20", "boto3>=1.34"]
|
|
40
|
+
vertex = ["opentelemetry-sdk>=1.20", "google-cloud-trace>=1.13"]
|
|
41
|
+
# live tracing of LangGraph apps: local mode ("file") and any OTLP collector
|
|
42
|
+
live = ["opentelemetry-sdk>=1.20", "opentelemetry-exporter-otlp-proto-http>=1.20",
|
|
43
|
+
"openinference-instrumentation-langchain>=0.1.20"]
|
|
44
|
+
# cloud mode (Cloud Run / VM -> Cloud Trace) and reading Cloud Trace for any mode, incl. Agent Engine
|
|
45
|
+
gcp = ["opentelemetry-sdk>=1.20", "opentelemetry-exporter-gcp-trace>=1.6", "google-cloud-trace>=1.13",
|
|
46
|
+
"openinference-instrumentation-langchain>=0.1.20"]
|
|
47
|
+
dev = ["pytest>=8", "langchain-core>=0.3", "langgraph>=0.2", "opentelemetry-sdk>=1.20"]
|
|
48
|
+
|
|
49
|
+
[project.scripts]
|
|
50
|
+
hieevas-dashboard = "hieevas.serve:main"
|
|
51
|
+
|
|
52
|
+
[project.urls]
|
|
53
|
+
Homepage = "https://github.com/heurius/hieevas"
|
|
54
|
+
Repository = "https://github.com/heurius/hieevas"
|
|
55
|
+
Issues = "https://github.com/heurius/hieevas/issues"
|
|
56
|
+
Documentation = "https://github.com/heurius/hieevas/blob/main/METRICS.md"
|
|
57
|
+
Changelog = "https://github.com/heurius/hieevas/blob/main/CHANGELOG.md"
|
|
58
|
+
|
|
59
|
+
[tool.setuptools.packages.find]
|
|
60
|
+
where = ["src"]
|
|
61
|
+
|
|
62
|
+
[tool.setuptools.package-data]
|
|
63
|
+
hieevas = ["py.typed"]
|
|
64
|
+
|
|
65
|
+
[tool.pytest.ini_options]
|
|
66
|
+
testpaths = ["tests"]
|
hieevas-0.1.0/setup.cfg
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""hieevas: multi-level evaluation of agentic AI and RAG pipelines.
|
|
2
|
+
|
|
3
|
+
Record what an agent does, then score it on 27 metrics across seven dimensions
|
|
4
|
+
(effectiveness, efficiency, planning, robustness, coordination, safety, governance)
|
|
5
|
+
at agent, interaction and system level.
|
|
6
|
+
|
|
7
|
+
from hieevas import Recorder, evaluate
|
|
8
|
+
rec = Recorder(app="my-rag", architecture="A1")
|
|
9
|
+
with rec.run("q1", reference="Paris") as run:
|
|
10
|
+
...
|
|
11
|
+
evaluate(rec.runs).to_html("dashboard.html")
|
|
12
|
+
|
|
13
|
+
Or trace a LangGraph app while it runs and view the dashboard (local, cloud or vertex mode):
|
|
14
|
+
|
|
15
|
+
session = hieevas.init("local") # or "cloud" / "vertex"
|
|
16
|
+
graph.invoke(inputs, config=hieevas.run_config(task_id="q1", reference="Paris"))
|
|
17
|
+
session.save_dashboard("dashboard.html") # or: python -m hieevas.serve --source ...
|
|
18
|
+
"""
|
|
19
|
+
from . import metrics
|
|
20
|
+
from .adapters.otel import annotate, run_config
|
|
21
|
+
from .evaluate import evaluate
|
|
22
|
+
from .governance import CANARY, governed_tool, inject, injection_followed
|
|
23
|
+
from .live import Session, init, live_report, load_runs, send_probes, setup_tracing
|
|
24
|
+
from .recorder import BudgetExceeded, Recorder, RunContext, current_run
|
|
25
|
+
from .report import Report
|
|
26
|
+
from .scoring import cohen_kappa, detect_refusal, score_answer, token_f1
|
|
27
|
+
from .stress import stress_tools
|
|
28
|
+
from .trace import RunRecord, Step, load_jsonl, save_csv, save_jsonl
|
|
29
|
+
|
|
30
|
+
__version__ = "0.1.0"
|
|
31
|
+
__all__ = ["Recorder", "RunContext", "current_run", "BudgetExceeded", "evaluate", "Report", "metrics",
|
|
32
|
+
"governed_tool", "inject", "injection_followed", "CANARY", "RunRecord", "Step", "save_jsonl",
|
|
33
|
+
"load_jsonl", "save_csv", "score_answer", "token_f1", "detect_refusal", "cohen_kappa",
|
|
34
|
+
"init", "Session", "setup_tracing", "load_runs", "live_report", "send_probes", "run_config", "annotate", "stress_tools"]
|