hieevas 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. hieevas-0.1.0/LICENSE +21 -0
  2. hieevas-0.1.0/PKG-INFO +262 -0
  3. hieevas-0.1.0/README.md +202 -0
  4. hieevas-0.1.0/pyproject.toml +66 -0
  5. hieevas-0.1.0/setup.cfg +4 -0
  6. hieevas-0.1.0/src/hieevas/__init__.py +34 -0
  7. hieevas-0.1.0/src/hieevas/adapters/__init__.py +2 -0
  8. hieevas-0.1.0/src/hieevas/adapters/langchain.py +103 -0
  9. hieevas-0.1.0/src/hieevas/adapters/otel.py +503 -0
  10. hieevas-0.1.0/src/hieevas/dashboard.py +323 -0
  11. hieevas-0.1.0/src/hieevas/evaluate.py +76 -0
  12. hieevas-0.1.0/src/hieevas/governance.py +127 -0
  13. hieevas-0.1.0/src/hieevas/live.py +322 -0
  14. hieevas-0.1.0/src/hieevas/metrics/__init__.py +23 -0
  15. hieevas-0.1.0/src/hieevas/metrics/base.py +83 -0
  16. hieevas-0.1.0/src/hieevas/metrics/definitions.py +375 -0
  17. hieevas-0.1.0/src/hieevas/metrics/guide.py +232 -0
  18. hieevas-0.1.0/src/hieevas/py.typed +0 -0
  19. hieevas-0.1.0/src/hieevas/recorder.py +175 -0
  20. hieevas-0.1.0/src/hieevas/report.py +120 -0
  21. hieevas-0.1.0/src/hieevas/scoring.py +144 -0
  22. hieevas-0.1.0/src/hieevas/serve.py +159 -0
  23. hieevas-0.1.0/src/hieevas/stress.py +77 -0
  24. hieevas-0.1.0/src/hieevas/trace.py +136 -0
  25. hieevas-0.1.0/src/hieevas.egg-info/PKG-INFO +262 -0
  26. hieevas-0.1.0/src/hieevas.egg-info/SOURCES.txt +35 -0
  27. hieevas-0.1.0/src/hieevas.egg-info/dependency_links.txt +1 -0
  28. hieevas-0.1.0/src/hieevas.egg-info/entry_points.txt +2 -0
  29. hieevas-0.1.0/src/hieevas.egg-info/requires.txt +38 -0
  30. hieevas-0.1.0/src/hieevas.egg-info/top_level.txt +1 -0
  31. hieevas-0.1.0/tests/test_langchain_adapter.py +62 -0
  32. hieevas-0.1.0/tests/test_live.py +176 -0
  33. hieevas-0.1.0/tests/test_metrics.py +127 -0
  34. hieevas-0.1.0/tests/test_otel_adapter.py +94 -0
  35. hieevas-0.1.0/tests/test_report_and_io.py +53 -0
  36. hieevas-0.1.0/tests/test_scoring.py +55 -0
  37. hieevas-0.1.0/tests/test_stress.py +97 -0
hieevas-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Sampathkumar T.
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
hieevas-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,262 @@
1
+ Metadata-Version: 2.4
2
+ Name: hieevas
3
+ Version: 0.1.0
4
+ Summary: Hierarchical evaluation of agentic systems (HIEEVAS): 27 process, cost, robustness, safety and governance metrics at agent, interaction and system level, computed from execution traces of LangGraph and OpenTelemetry-instrumented agents.
5
+ Author-email: "Sampathkumar T." <heurius.tech@gmail.com>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/heurius/hieevas
8
+ Project-URL: Repository, https://github.com/heurius/hieevas
9
+ Project-URL: Issues, https://github.com/heurius/hieevas/issues
10
+ Project-URL: Documentation, https://github.com/heurius/hieevas/blob/main/METRICS.md
11
+ Project-URL: Changelog, https://github.com/heurius/hieevas/blob/main/CHANGELOG.md
12
+ Keywords: agentic-ai,ai-agents,llm,evaluation,metrics,langgraph,langchain,rag,opentelemetry,observability,ai-governance,prompt-injection,vertex-ai
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Topic :: Software Development :: Testing
25
+ Classifier: Topic :: System :: Monitoring
26
+ Classifier: Typing :: Typed
27
+ Requires-Python: >=3.10
28
+ Description-Content-Type: text/markdown
29
+ License-File: LICENSE
30
+ Provides-Extra: langchain
31
+ Requires-Dist: langchain-core>=0.3; extra == "langchain"
32
+ Provides-Extra: langgraph
33
+ Requires-Dist: langchain-core>=0.3; extra == "langgraph"
34
+ Requires-Dist: langgraph>=0.2; extra == "langgraph"
35
+ Provides-Extra: pandas
36
+ Requires-Dist: pandas>=2.0; extra == "pandas"
37
+ Provides-Extra: otel
38
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "otel"
39
+ Provides-Extra: agentcore
40
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "agentcore"
41
+ Requires-Dist: boto3>=1.34; extra == "agentcore"
42
+ Provides-Extra: vertex
43
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "vertex"
44
+ Requires-Dist: google-cloud-trace>=1.13; extra == "vertex"
45
+ Provides-Extra: live
46
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "live"
47
+ Requires-Dist: opentelemetry-exporter-otlp-proto-http>=1.20; extra == "live"
48
+ Requires-Dist: openinference-instrumentation-langchain>=0.1.20; extra == "live"
49
+ Provides-Extra: gcp
50
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "gcp"
51
+ Requires-Dist: opentelemetry-exporter-gcp-trace>=1.6; extra == "gcp"
52
+ Requires-Dist: google-cloud-trace>=1.13; extra == "gcp"
53
+ Requires-Dist: openinference-instrumentation-langchain>=0.1.20; extra == "gcp"
54
+ Provides-Extra: dev
55
+ Requires-Dist: pytest>=8; extra == "dev"
56
+ Requires-Dist: langchain-core>=0.3; extra == "dev"
57
+ Requires-Dist: langgraph>=0.2; extra == "dev"
58
+ Requires-Dist: opentelemetry-sdk>=1.20; extra == "dev"
59
+ Dynamic: license-file
60
+
61
+ # hieevas
62
+
63
+ **HIE**rarchical **EV**aluation of **A**gentic **S**ystems: evaluate **how** an agentic AI or RAG
64
+ pipeline reaches its answers, not only **whether** it does.
65
+
66
+ Answer-quality tools such as Ragas judge the final answer and its grounding; hieevas measures the
67
+ path to it (cost, robustness under tool faults, prompt-injection resistance, governance and audit)
68
+ at three levels of the hierarchy: the single agent, the interaction between agents, and the whole
69
+ system. The two complement each other.
70
+
71
+ `hieevas` records what your pipeline does (LLM calls, tool calls, messages between agents,
72
+ reviews, approvals) and scores it on **27 metrics in 7 dimensions** at **agent, interaction and
73
+ system level**, following the multi-level evaluation framework in *Measuring the Process, Not Just
74
+ the Outcome* (Table 8). It works offline, has no required dependencies, and produces a
75
+ self-contained HTML dashboard.
76
+
77
+ | Dimension | Metrics |
78
+ |---|---|
79
+ | Effectiveness | M1 task success · M2 answer F1 · M3 consistency |
80
+ | Efficiency | M4 completion time · M5 tokens/task · M6 LLM calls · M7 tokens per success · M8 peak context |
81
+ | Planning & reasoning | M9 steps · M10 redundant calls · M11 plan adherence · M12 tool-call validity |
82
+ | Robustness & recovery | M13 fault-induced drop · M14 recovery rate · M15 retries per error |
83
+ | Coordination | M16 hand-over success · M17 reviewer rejections · M18 correction success · M19 communication overhead |
84
+ | Safety & security | M20 refusal rate · M21 injection success · M22 unauthorised high-risk attempts |
85
+ | Transparency & governance | M23 rationale coverage · M24 rationale quality · M25 audit completeness · M26 approval triggers · M27 budget violations |
86
+
87
+ When the data for a metric was not recorded, the metric returns *not available* with the reason,
88
+ never a misleading zero. [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md) explains every metric in plain language: what it
89
+ measures, how it is computed, and which part of a LangGraph run it reads (input, LLM call, tool call,
90
+ final state). The same text appears under *What is this?* on each dashboard card.
91
+
92
+ ## Install
93
+
94
+ ```bash
95
+ pip install hieevas # core, no dependencies
96
+ pip install "hieevas[langgraph]" # + LangChain / LangGraph callback adapter
97
+ pip install "hieevas[live]" # + live tracing of LangGraph apps (local mode, any OTLP collector)
98
+ pip install "hieevas[gcp]" # + Cloud Run / VM -> Cloud Trace, and reading Cloud Trace (incl. Agent Engine)
99
+ ```
100
+
101
+ Until the first PyPI release, install from GitHub:
102
+ `pip install "hieevas[live] @ git+https://github.com/heurius/hieevas"`.
103
+ For development: `git clone https://github.com/heurius/hieevas && pip install -e "./hieevas[dev]"`.
104
+
105
+ ## 0. Live dashboards for any LangGraph app
106
+
107
+ Add one line at start-up; tag requests only if you want known-answer or stress probes.
108
+
109
+ ```python
110
+ import hieevas
111
+
112
+ session = hieevas.init("local") # or "cloud" (Cloud Run / VM) or "vertex" (Agent Engine)
113
+ graph.invoke(inputs) # live traffic: no changes
114
+ graph.invoke(inputs, config=hieevas.run_config(task_id="q1", condition="C1", reference="Paris")) # probe
115
+ session.save_dashboard("dashboard.html")
116
+ ```
117
+
118
+ ```bash
119
+ python -m hieevas.serve --source file:traces/spans.jsonl # local mode
120
+ python -m hieevas.serve --source cloud-trace:MY_PROJECT --tool-risk send_email # cloud and vertex modes
121
+ ```
122
+
123
+ | Mode | App and LLM | Traces written by | Dashboard reads |
124
+ |---|---|---|---|
125
+ | `local` | both on this machine (e.g. Ollama) | `hieevas.init("local")` → JSON-lines file | `file:traces/spans.jsonl` |
126
+ | `cloud` | app on Cloud Run / a VM, LLM via an API | `hieevas.init("cloud")` → Cloud Trace (or `endpoint=` any OTLP collector) | `cloud-trace:PROJECT` |
127
+ | `vertex` | agent on Vertex AI Agent Engine | `LanggraphAgent(..., enable_tracing=True)` → Cloud Trace | `cloud-trace:PROJECT` |
128
+
129
+ All three produce the same OpenInference spans, so one reader and one dashboard serve them.
130
+ Tags passed with `run_config` travel in the LangGraph config metadata and are read back
131
+ from the spans, which also works on Agent Engine where the app does not own its spans.
132
+ The dashboard server (`/`, `/report.json`, `/healthz`; query `?hours=6&group_by=architecture,hour`)
133
+ needs only the standard library; on Cloud Run deploy it with `--no-allow-unauthenticated`.
134
+
135
+ **What live traffic can and cannot show.** Real user requests have no reference answers, so
136
+ they yield the efficiency, planning, coordination, high-risk-attempt and governance metrics.
137
+ Task success, recovery, refusal and injection metrics come from probes: requests tagged with a
138
+ condition (`C1` known answer, `C2` tool faults, `C3` injection, `C4` out-of-policy) sent with
139
+ `hieevas.send_probes(...)`, for example on a schedule. For `C4` runs without an explicit
140
+ flag, refusal is scored as *declined in words and no high-risk tool attempted*.
141
+
142
+ Examples for each mode are in [`examples/`](https://github.com/heurius/hieevas/tree/main/examples): `local_ollama.py`, `cloud_run/`,
143
+ `vertex_agent_engine.py`, and a stand-alone dashboard container in `dashboard/`.
144
+
145
+ ## Using hieevas in production
146
+
147
+ hieevas is alpha software, provided "as is" under the MIT licence without warranty. Before
148
+ running it next to real users, check these four points:
149
+
150
+ 1. **The approval gate blocks real actions by default.** `stress_tools(..., high_risk=[...])`
151
+ routes every call to a high-risk tool through `approval`, for probes *and* real users, and the
152
+ default `deny_all` refuses them all. In production pass your own approval function (for example
153
+ one that asks a person), or `approval=approve_all` from `hieevas.governance` to record without
154
+ blocking.
155
+ 2. **Stress conditions are for probes only.** Tool faults (`C2`) and injected text (`C3`) are
156
+ applied only to requests tagged with that condition. Never let end users set tags: the Cloud
157
+ Run example rejects tags unless `HIEEVAS_ACCEPT_TAGS=1`, and probes should come from a
158
+ trusted caller.
159
+ 3. **Traces contain user data.** Spans hold prompts, retrieved passages, tool arguments and
160
+ answers, which may include personal or confidential information. Treat trace files and Cloud
161
+ Trace like application logs: restrict access, set retention, sample (`sample_rate=`) and
162
+ follow your privacy obligations. The dashboard server has no login of its own; locally it
163
+ listens on 127.0.0.1 only, and on Cloud Run it must be deployed with
164
+ `--no-allow-unauthenticated` or behind IAP.
165
+ 4. **Existing OpenTelemetry setups are joined, not replaced.** If the application already
166
+ configured a tracer provider, `hieevas.init()` adds its exporter to that provider and warns;
167
+ the application's service name and sampling stay in force.
168
+
169
+ Metric values are measurements from recorded behaviour, not certifications: keyword-based
170
+ refusal detection and phrase-matching of answers are heuristics (see [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md)).
171
+ See [SECURITY.md](https://github.com/heurius/hieevas/blob/main/SECURITY.md) to report a vulnerability.
172
+
173
+ ## 1. Any pipeline (framework-agnostic)
174
+
175
+ ```python
176
+ from hieevas import Recorder, evaluate
177
+
178
+ rec = Recorder(app="my-rag", architecture="A1", model="qwen2.5:1.5b")
179
+ with rec.run("q1", reference="Paris") as run:
180
+ run.llm_call(agent="retriever", input_tokens=120, output_tokens=18, latency=0.9)
181
+ run.tool_call("search_docs", {"query": "capital of France"}, rationale="need a source")
182
+ run.answer("Paris")
183
+
184
+ report = evaluate(rec.runs, group_by=("architecture", "condition"))
185
+ print(report.summary())
186
+ report.to_html("dashboard.html") # offline dashboard
187
+ report.to_csv("metrics.csv")
188
+ rec.save("traces.jsonl"); rec.save_csv("logs/") # runs.csv + steps.csv
189
+ ```
190
+
191
+ Multi-agent pipelines can also record `run.plan([...])`, `run.plan_step_done(i)`,
192
+ `run.message(sender, receiver, text)`, `run.review("accept"|"reject")` and
193
+ `run.mark(refused=..., injection_followed=...)`.
194
+
195
+ ## 2. LangChain / LangGraph
196
+
197
+ ```python
198
+ from hieevas.adapters.langchain import HieevasCallbackHandler
199
+
200
+ with rec.run("q1", reference="Paris") as run:
201
+ out = graph.invoke(inputs, config={"callbacks": [HieevasCallbackHandler(run)],
202
+ "recursion_limit": 40}) # recursion limit = step budget (M27)
203
+ run.answer(out["messages"][-1].content)
204
+ ```
205
+
206
+ The handler records every LLM call (tokens, latency, agent = LangGraph node) and every tool call
207
+ (status, arguments, and the text the model wrote before calling it, used as the rationale).
208
+
209
+ ## 3. Governance layer and test conditions
210
+
211
+ ```python
212
+ from hieevas import governed_tool, inject, injection_followed
213
+
214
+ search = governed_tool(search_fn, name="search_docs", fault_rate=0.3) # C2 tool faults
215
+ send_email = governed_tool(send_fn, name="send_email", risk="high") # approval gate (denies by default)
216
+ docs = inject(retrieved_docs) # C3 prompt injection
217
+ run.mark(injection_followed=injection_followed(run.record))
218
+ ```
219
+
220
+ ## 4. OpenTelemetry: AgentCore, Vertex AI and any OTel backend
221
+
222
+ One trace becomes one run. Spans using the OpenTelemetry GenAI conventions (`gen_ai.*`),
223
+ OpenInference or OpenLLMetry are understood.
224
+
225
+ ```python
226
+ # a) In-process, wherever the agent runs (locally, AgentCore Runtime, Vertex Agent Engine)
227
+ from hieevas.adapters.otel import HieevasSpanProcessor
228
+ proc = HieevasSpanProcessor()
229
+ tracer_provider.add_span_processor(proc) # alongside your normal exporter
230
+ ... # run the agent
231
+ report = evaluate(proc.runs(architecture="A1"))
232
+
233
+ # b) From files written by an OpenTelemetry Collector (file exporter)
234
+ from hieevas.adapters.otel import load_otlp_json, spans_to_runs
235
+ runs = spans_to_runs(load_otlp_json("traces.json"))
236
+
237
+ # c) Amazon Bedrock AgentCore: spans in CloudWatch (log group aws/spans) pip install "hieevas[agentcore]"
238
+ from hieevas.adapters.otel import fetch_cloudwatch_spans
239
+ runs = spans_to_runs(fetch_cloudwatch_spans(start_ms, end_ms, region="us-east-1"))
240
+
241
+ # d) Google Vertex AI Agent Engine: spans in Cloud Trace pip install "hieevas[vertex]"
242
+ from hieevas.adapters.otel import fetch_cloud_trace_spans
243
+ runs = spans_to_runs(fetch_cloud_trace_spans("my-gcp-project"))
244
+ ```
245
+
246
+ Set `hieevas.task_id`, `hieevas.condition`, `hieevas.reference` and `hieevas.architecture`
247
+ as span attributes to control grouping and scoring; `hieevas.rationale` and `hieevas.risk`
248
+ on tool spans enable M23 and M22. The CloudWatch and Cloud Trace loaders are experimental:
249
+ check the span record format for your platform version.
250
+
251
+ ## Example
252
+
253
+ `../rag_eval/` runs a RAG pipeline over four PDF papers with a local Ollama model in three
254
+ architectures (single agent, planner–executor, planner–executor–reviewer, built with LangGraph)
255
+ under four conditions, and writes the dashboard.
256
+
257
+ ## Notes
258
+
259
+ - M20 and M24 are rubric-scored. `detect_refusal` is a keyword heuristic for screening only;
260
+ confirm codes with human raters and report agreement with `cohen_kappa`.
261
+ - Token counts come from the model provider; message tokens are estimated (4 characters ≈ 1 token)
262
+ when not supplied.
@@ -0,0 +1,202 @@
1
+ # hieevas
2
+
3
+ **HIE**rarchical **EV**aluation of **A**gentic **S**ystems: evaluate **how** an agentic AI or RAG
4
+ pipeline reaches its answers, not only **whether** it does.
5
+
6
+ Answer-quality tools such as Ragas judge the final answer and its grounding; hieevas measures the
7
+ path to it (cost, robustness under tool faults, prompt-injection resistance, governance and audit)
8
+ at three levels of the hierarchy: the single agent, the interaction between agents, and the whole
9
+ system. The two complement each other.
10
+
11
+ `hieevas` records what your pipeline does (LLM calls, tool calls, messages between agents,
12
+ reviews, approvals) and scores it on **27 metrics in 7 dimensions** at **agent, interaction and
13
+ system level**, following the multi-level evaluation framework in *Measuring the Process, Not Just
14
+ the Outcome* (Table 8). It works offline, has no required dependencies, and produces a
15
+ self-contained HTML dashboard.
16
+
17
+ | Dimension | Metrics |
18
+ |---|---|
19
+ | Effectiveness | M1 task success · M2 answer F1 · M3 consistency |
20
+ | Efficiency | M4 completion time · M5 tokens/task · M6 LLM calls · M7 tokens per success · M8 peak context |
21
+ | Planning & reasoning | M9 steps · M10 redundant calls · M11 plan adherence · M12 tool-call validity |
22
+ | Robustness & recovery | M13 fault-induced drop · M14 recovery rate · M15 retries per error |
23
+ | Coordination | M16 hand-over success · M17 reviewer rejections · M18 correction success · M19 communication overhead |
24
+ | Safety & security | M20 refusal rate · M21 injection success · M22 unauthorised high-risk attempts |
25
+ | Transparency & governance | M23 rationale coverage · M24 rationale quality · M25 audit completeness · M26 approval triggers · M27 budget violations |
26
+
27
+ When the data for a metric was not recorded, the metric returns *not available* with the reason,
28
+ never a misleading zero. [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md) explains every metric in plain language: what it
29
+ measures, how it is computed, and which part of a LangGraph run it reads (input, LLM call, tool call,
30
+ final state). The same text appears under *What is this?* on each dashboard card.
31
+
32
+ ## Install
33
+
34
+ ```bash
35
+ pip install hieevas # core, no dependencies
36
+ pip install "hieevas[langgraph]" # + LangChain / LangGraph callback adapter
37
+ pip install "hieevas[live]" # + live tracing of LangGraph apps (local mode, any OTLP collector)
38
+ pip install "hieevas[gcp]" # + Cloud Run / VM -> Cloud Trace, and reading Cloud Trace (incl. Agent Engine)
39
+ ```
40
+
41
+ Until the first PyPI release, install from GitHub:
42
+ `pip install "hieevas[live] @ git+https://github.com/heurius/hieevas"`.
43
+ For development: `git clone https://github.com/heurius/hieevas && pip install -e "./hieevas[dev]"`.
44
+
45
+ ## 0. Live dashboards for any LangGraph app
46
+
47
+ Add one line at start-up; tag requests only if you want known-answer or stress probes.
48
+
49
+ ```python
50
+ import hieevas
51
+
52
+ session = hieevas.init("local") # or "cloud" (Cloud Run / VM) or "vertex" (Agent Engine)
53
+ graph.invoke(inputs) # live traffic: no changes
54
+ graph.invoke(inputs, config=hieevas.run_config(task_id="q1", condition="C1", reference="Paris")) # probe
55
+ session.save_dashboard("dashboard.html")
56
+ ```
57
+
58
+ ```bash
59
+ python -m hieevas.serve --source file:traces/spans.jsonl # local mode
60
+ python -m hieevas.serve --source cloud-trace:MY_PROJECT --tool-risk send_email # cloud and vertex modes
61
+ ```
62
+
63
+ | Mode | App and LLM | Traces written by | Dashboard reads |
64
+ |---|---|---|---|
65
+ | `local` | both on this machine (e.g. Ollama) | `hieevas.init("local")` → JSON-lines file | `file:traces/spans.jsonl` |
66
+ | `cloud` | app on Cloud Run / a VM, LLM via an API | `hieevas.init("cloud")` → Cloud Trace (or `endpoint=` any OTLP collector) | `cloud-trace:PROJECT` |
67
+ | `vertex` | agent on Vertex AI Agent Engine | `LanggraphAgent(..., enable_tracing=True)` → Cloud Trace | `cloud-trace:PROJECT` |
68
+
69
+ All three produce the same OpenInference spans, so one reader and one dashboard serve them.
70
+ Tags passed with `run_config` travel in the LangGraph config metadata and are read back
71
+ from the spans, which also works on Agent Engine where the app does not own its spans.
72
+ The dashboard server (`/`, `/report.json`, `/healthz`; query `?hours=6&group_by=architecture,hour`)
73
+ needs only the standard library; on Cloud Run deploy it with `--no-allow-unauthenticated`.
74
+
75
+ **What live traffic can and cannot show.** Real user requests have no reference answers, so
76
+ they yield the efficiency, planning, coordination, high-risk-attempt and governance metrics.
77
+ Task success, recovery, refusal and injection metrics come from probes: requests tagged with a
78
+ condition (`C1` known answer, `C2` tool faults, `C3` injection, `C4` out-of-policy) sent with
79
+ `hieevas.send_probes(...)`, for example on a schedule. For `C4` runs without an explicit
80
+ flag, refusal is scored as *declined in words and no high-risk tool attempted*.
81
+
82
+ Examples for each mode are in [`examples/`](https://github.com/heurius/hieevas/tree/main/examples): `local_ollama.py`, `cloud_run/`,
83
+ `vertex_agent_engine.py`, and a stand-alone dashboard container in `dashboard/`.
84
+
85
+ ## Using hieevas in production
86
+
87
+ hieevas is alpha software, provided "as is" under the MIT licence without warranty. Before
88
+ running it next to real users, check these four points:
89
+
90
+ 1. **The approval gate blocks real actions by default.** `stress_tools(..., high_risk=[...])`
91
+ routes every call to a high-risk tool through `approval`, for probes *and* real users, and the
92
+ default `deny_all` refuses them all. In production pass your own approval function (for example
93
+ one that asks a person), or `approval=approve_all` from `hieevas.governance` to record without
94
+ blocking.
95
+ 2. **Stress conditions are for probes only.** Tool faults (`C2`) and injected text (`C3`) are
96
+ applied only to requests tagged with that condition. Never let end users set tags: the Cloud
97
+ Run example rejects tags unless `HIEEVAS_ACCEPT_TAGS=1`, and probes should come from a
98
+ trusted caller.
99
+ 3. **Traces contain user data.** Spans hold prompts, retrieved passages, tool arguments and
100
+ answers, which may include personal or confidential information. Treat trace files and Cloud
101
+ Trace like application logs: restrict access, set retention, sample (`sample_rate=`) and
102
+ follow your privacy obligations. The dashboard server has no login of its own; locally it
103
+ listens on 127.0.0.1 only, and on Cloud Run it must be deployed with
104
+ `--no-allow-unauthenticated` or behind IAP.
105
+ 4. **Existing OpenTelemetry setups are joined, not replaced.** If the application already
106
+ configured a tracer provider, `hieevas.init()` adds its exporter to that provider and warns;
107
+ the application's service name and sampling stay in force.
108
+
109
+ Metric values are measurements from recorded behaviour, not certifications: keyword-based
110
+ refusal detection and phrase-matching of answers are heuristics (see [METRICS.md](https://github.com/heurius/hieevas/blob/main/METRICS.md)).
111
+ See [SECURITY.md](https://github.com/heurius/hieevas/blob/main/SECURITY.md) to report a vulnerability.
112
+
113
+ ## 1. Any pipeline (framework-agnostic)
114
+
115
+ ```python
116
+ from hieevas import Recorder, evaluate
117
+
118
+ rec = Recorder(app="my-rag", architecture="A1", model="qwen2.5:1.5b")
119
+ with rec.run("q1", reference="Paris") as run:
120
+ run.llm_call(agent="retriever", input_tokens=120, output_tokens=18, latency=0.9)
121
+ run.tool_call("search_docs", {"query": "capital of France"}, rationale="need a source")
122
+ run.answer("Paris")
123
+
124
+ report = evaluate(rec.runs, group_by=("architecture", "condition"))
125
+ print(report.summary())
126
+ report.to_html("dashboard.html") # offline dashboard
127
+ report.to_csv("metrics.csv")
128
+ rec.save("traces.jsonl"); rec.save_csv("logs/") # runs.csv + steps.csv
129
+ ```
130
+
131
+ Multi-agent pipelines can also record `run.plan([...])`, `run.plan_step_done(i)`,
132
+ `run.message(sender, receiver, text)`, `run.review("accept"|"reject")` and
133
+ `run.mark(refused=..., injection_followed=...)`.
134
+
135
+ ## 2. LangChain / LangGraph
136
+
137
+ ```python
138
+ from hieevas.adapters.langchain import HieevasCallbackHandler
139
+
140
+ with rec.run("q1", reference="Paris") as run:
141
+ out = graph.invoke(inputs, config={"callbacks": [HieevasCallbackHandler(run)],
142
+ "recursion_limit": 40}) # recursion limit = step budget (M27)
143
+ run.answer(out["messages"][-1].content)
144
+ ```
145
+
146
+ The handler records every LLM call (tokens, latency, agent = LangGraph node) and every tool call
147
+ (status, arguments, and the text the model wrote before calling it, used as the rationale).
148
+
149
+ ## 3. Governance layer and test conditions
150
+
151
+ ```python
152
+ from hieevas import governed_tool, inject, injection_followed
153
+
154
+ search = governed_tool(search_fn, name="search_docs", fault_rate=0.3) # C2 tool faults
155
+ send_email = governed_tool(send_fn, name="send_email", risk="high") # approval gate (denies by default)
156
+ docs = inject(retrieved_docs) # C3 prompt injection
157
+ run.mark(injection_followed=injection_followed(run.record))
158
+ ```
159
+
160
+ ## 4. OpenTelemetry: AgentCore, Vertex AI and any OTel backend
161
+
162
+ One trace becomes one run. Spans using the OpenTelemetry GenAI conventions (`gen_ai.*`),
163
+ OpenInference or OpenLLMetry are understood.
164
+
165
+ ```python
166
+ # a) In-process, wherever the agent runs (locally, AgentCore Runtime, Vertex Agent Engine)
167
+ from hieevas.adapters.otel import HieevasSpanProcessor
168
+ proc = HieevasSpanProcessor()
169
+ tracer_provider.add_span_processor(proc) # alongside your normal exporter
170
+ ... # run the agent
171
+ report = evaluate(proc.runs(architecture="A1"))
172
+
173
+ # b) From files written by an OpenTelemetry Collector (file exporter)
174
+ from hieevas.adapters.otel import load_otlp_json, spans_to_runs
175
+ runs = spans_to_runs(load_otlp_json("traces.json"))
176
+
177
+ # c) Amazon Bedrock AgentCore: spans in CloudWatch (log group aws/spans) pip install "hieevas[agentcore]"
178
+ from hieevas.adapters.otel import fetch_cloudwatch_spans
179
+ runs = spans_to_runs(fetch_cloudwatch_spans(start_ms, end_ms, region="us-east-1"))
180
+
181
+ # d) Google Vertex AI Agent Engine: spans in Cloud Trace pip install "hieevas[vertex]"
182
+ from hieevas.adapters.otel import fetch_cloud_trace_spans
183
+ runs = spans_to_runs(fetch_cloud_trace_spans("my-gcp-project"))
184
+ ```
185
+
186
+ Set `hieevas.task_id`, `hieevas.condition`, `hieevas.reference` and `hieevas.architecture`
187
+ as span attributes to control grouping and scoring; `hieevas.rationale` and `hieevas.risk`
188
+ on tool spans enable M23 and M22. The CloudWatch and Cloud Trace loaders are experimental:
189
+ check the span record format for your platform version.
190
+
191
+ ## Example
192
+
193
+ `../rag_eval/` runs a RAG pipeline over four PDF papers with a local Ollama model in three
194
+ architectures (single agent, planner–executor, planner–executor–reviewer, built with LangGraph)
195
+ under four conditions, and writes the dashboard.
196
+
197
+ ## Notes
198
+
199
+ - M20 and M24 are rubric-scored. `detect_refusal` is a keyword heuristic for screening only;
200
+ confirm codes with human raters and report agreement with `cohen_kappa`.
201
+ - Token counts come from the model provider; message tokens are estimated (4 characters ≈ 1 token)
202
+ when not supplied.
@@ -0,0 +1,66 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "hieevas"
7
+ version = "0.1.0"
8
+ description = "Hierarchical evaluation of agentic systems (HIEEVAS): 27 process, cost, robustness, safety and governance metrics at agent, interaction and system level, computed from execution traces of LangGraph and OpenTelemetry-instrumented agents."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ authors = [{ name = "Sampathkumar T.", email = "heurius.tech@gmail.com" }]
14
+ keywords = ["agentic-ai", "ai-agents", "llm", "evaluation", "metrics", "langgraph", "langchain", "rag",
15
+ "opentelemetry", "observability", "ai-governance", "prompt-injection", "vertex-ai"]
16
+ classifiers = [
17
+ "Development Status :: 3 - Alpha",
18
+ "Intended Audience :: Developers",
19
+ "Intended Audience :: Science/Research",
20
+ "Operating System :: OS Independent",
21
+ "Programming Language :: Python :: 3",
22
+ "Programming Language :: Python :: 3 :: Only",
23
+ "Programming Language :: Python :: 3.10",
24
+ "Programming Language :: Python :: 3.11",
25
+ "Programming Language :: Python :: 3.12",
26
+ "Programming Language :: Python :: 3.13",
27
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
28
+ "Topic :: Software Development :: Testing",
29
+ "Topic :: System :: Monitoring",
30
+ "Typing :: Typed",
31
+ ]
32
+ dependencies = []
33
+
34
+ [project.optional-dependencies]
35
+ langchain = ["langchain-core>=0.3"]
36
+ langgraph = ["langchain-core>=0.3", "langgraph>=0.2"]
37
+ pandas = ["pandas>=2.0"]
38
+ otel = ["opentelemetry-sdk>=1.20"]
39
+ agentcore = ["opentelemetry-sdk>=1.20", "boto3>=1.34"]
40
+ vertex = ["opentelemetry-sdk>=1.20", "google-cloud-trace>=1.13"]
41
+ # live tracing of LangGraph apps: local mode ("file") and any OTLP collector
42
+ live = ["opentelemetry-sdk>=1.20", "opentelemetry-exporter-otlp-proto-http>=1.20",
43
+ "openinference-instrumentation-langchain>=0.1.20"]
44
+ # cloud mode (Cloud Run / VM -> Cloud Trace) and reading Cloud Trace for any mode, incl. Agent Engine
45
+ gcp = ["opentelemetry-sdk>=1.20", "opentelemetry-exporter-gcp-trace>=1.6", "google-cloud-trace>=1.13",
46
+ "openinference-instrumentation-langchain>=0.1.20"]
47
+ dev = ["pytest>=8", "langchain-core>=0.3", "langgraph>=0.2", "opentelemetry-sdk>=1.20"]
48
+
49
+ [project.scripts]
50
+ hieevas-dashboard = "hieevas.serve:main"
51
+
52
+ [project.urls]
53
+ Homepage = "https://github.com/heurius/hieevas"
54
+ Repository = "https://github.com/heurius/hieevas"
55
+ Issues = "https://github.com/heurius/hieevas/issues"
56
+ Documentation = "https://github.com/heurius/hieevas/blob/main/METRICS.md"
57
+ Changelog = "https://github.com/heurius/hieevas/blob/main/CHANGELOG.md"
58
+
59
+ [tool.setuptools.packages.find]
60
+ where = ["src"]
61
+
62
+ [tool.setuptools.package-data]
63
+ hieevas = ["py.typed"]
64
+
65
+ [tool.pytest.ini_options]
66
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,34 @@
1
+ """hieevas: multi-level evaluation of agentic AI and RAG pipelines.
2
+
3
+ Record what an agent does, then score it on 27 metrics across seven dimensions
4
+ (effectiveness, efficiency, planning, robustness, coordination, safety, governance)
5
+ at agent, interaction and system level.
6
+
7
+ from hieevas import Recorder, evaluate
8
+ rec = Recorder(app="my-rag", architecture="A1")
9
+ with rec.run("q1", reference="Paris") as run:
10
+ ...
11
+ evaluate(rec.runs).to_html("dashboard.html")
12
+
13
+ Or trace a LangGraph app while it runs and view the dashboard (local, cloud or vertex mode):
14
+
15
+ session = hieevas.init("local") # or "cloud" / "vertex"
16
+ graph.invoke(inputs, config=hieevas.run_config(task_id="q1", reference="Paris"))
17
+ session.save_dashboard("dashboard.html") # or: python -m hieevas.serve --source ...
18
+ """
19
+ from . import metrics
20
+ from .adapters.otel import annotate, run_config
21
+ from .evaluate import evaluate
22
+ from .governance import CANARY, governed_tool, inject, injection_followed
23
+ from .live import Session, init, live_report, load_runs, send_probes, setup_tracing
24
+ from .recorder import BudgetExceeded, Recorder, RunContext, current_run
25
+ from .report import Report
26
+ from .scoring import cohen_kappa, detect_refusal, score_answer, token_f1
27
+ from .stress import stress_tools
28
+ from .trace import RunRecord, Step, load_jsonl, save_csv, save_jsonl
29
+
30
+ __version__ = "0.1.0"
31
+ __all__ = ["Recorder", "RunContext", "current_run", "BudgetExceeded", "evaluate", "Report", "metrics",
32
+ "governed_tool", "inject", "injection_followed", "CANARY", "RunRecord", "Step", "save_jsonl",
33
+ "load_jsonl", "save_csv", "score_answer", "token_f1", "detect_refusal", "cohen_kappa",
34
+ "init", "Session", "setup_tracing", "load_runs", "live_report", "send_probes", "run_config", "annotate", "stress_tools"]
@@ -0,0 +1,2 @@
1
+ """Framework adapters. Import the one you need, e.g.
2
+ ``from hieevas.adapters.langchain import HieevasCallbackHandler``."""