agenthawk 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agenthawk-0.2.0 → agenthawk-0.2.1}/PKG-INFO +13 -24
- {agenthawk-0.2.0 → agenthawk-0.2.1}/README.md +13 -24
- {agenthawk-0.2.0 → agenthawk-0.2.1}/pyproject.toml +1 -1
- {agenthawk-0.2.0 → agenthawk-0.2.1}/server.json +2 -2
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/mcp_trace/__init__.py +1 -1
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/mcp_trace/core.py +7 -2
- {agenthawk-0.2.0 → agenthawk-0.2.1}/tests/test_core.py +9 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/.github/workflows/ci.yml +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/.github/workflows/publish-pypi.yml +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/.gitignore +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/CONTRIBUTING.md +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/LICENSE +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/examples/example_trace.jsonl +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/agenthawk/__init__.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/agenthawk/__main__.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/agenthawk/core.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/agenthawk/server.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/mcp_trace/__main__.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/src/mcp_trace/server.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/tests/conftest.py +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/tests/fixtures/demo_trace.jsonl +0 -0
- {agenthawk-0.2.0 → agenthawk-0.2.1}/tests/test_server.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: agenthawk
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Agent observability MCP server for querying OpenTelemetry traces, failures, tool health, security evidence, and regressions.
|
|
5
5
|
Project-URL: Homepage, https://github.com/abhishekash/agenthawk
|
|
6
6
|
Project-URL: Repository, https://github.com/abhishekash/agenthawk
|
|
@@ -26,11 +26,17 @@ Description-Content-Type: text/markdown
|
|
|
26
26
|
|
|
27
27
|
[](https://github.com/abhishekash/agenthawk/actions/workflows/ci.yml) [](LICENSE)
|
|
28
28
|
|
|
29
|
-
|
|
29
|
+
AgentHawk is a small, local-first stdio MCP server for querying JSONL
|
|
30
|
+
OpenTelemetry spans emitted by agent runs. It exposes focused queries for
|
|
31
|
+
run summaries, tool failures, approvals, activity, and comparisons instead of
|
|
32
|
+
requiring an application-specific dashboard.
|
|
30
33
|
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
+
It pairs with [agent-harness](https://github.com/abhishekash/agent-harness),
|
|
35
|
+
but the reader is format-simple: any JSONL of OTel-shaped spans works. The
|
|
36
|
+
project's concrete origin was a shell-boundary regression: the harness blocked
|
|
37
|
+
`cat ../outside.txt`, but the old trace recorded only that `run_shell` was
|
|
38
|
+
called, not the returned diagnostic. AgentHawk and the harness now preserve and
|
|
39
|
+
query that bounded error.
|
|
34
40
|
|
|
35
41
|
## AI-native use cases
|
|
36
42
|
|
|
@@ -58,7 +64,7 @@ cd agenthawk && uv pip install -e .
|
|
|
58
64
|
agenthawk --trace-dir ./traces
|
|
59
65
|
```
|
|
60
66
|
|
|
61
|
-
The `agenthawk` 0.2.
|
|
67
|
+
The `agenthawk` 0.2.1 release is the renamed successor to the published
|
|
62
68
|
[`abhishekash-mcp-trace` 0.1.1](https://pypi.org/project/abhishekash-mcp-trace/0.1.1/)
|
|
63
69
|
distribution. Existing clients can continue using the legacy command while
|
|
64
70
|
migrating.
|
|
@@ -141,27 +147,10 @@ traces/*.jsonl ──▶ agenthawk.core (pure query functions, zero deps)
|
|
|
141
147
|
- **core/server split**: all logic is pure functions over parsed spans; the MCP layer only parses args and JSON-encodes results. Tests hit both layers.
|
|
142
148
|
- **trace_id prefixes**: agents fumble full 32-char hex ids; every tool accepts prefixes.
|
|
143
149
|
- **bounded output**: diagnostics are truncated and obvious credentials are redacted before query results leave the server.
|
|
150
|
+
- **honest cost totals**: `token_usage` returns `cost_known: false` and a null cost when any producer marks pricing as unavailable; unknown pricing is never presented as `$0`.
|
|
144
151
|
- **cursor polling**: `recent_activity` makes the snapshot reader useful while a run is still writing spans.
|
|
145
152
|
- The demo fixture ([`examples/example_trace.jsonl`](examples/example_trace.jsonl)) is a *real* agent-harness run, not hand-written.
|
|
146
153
|
|
|
147
|
-
## Research-driven gaps addressed
|
|
148
|
-
|
|
149
|
-
A small Reddit review surfaced the same production problems repeatedly: auth and
|
|
150
|
-
identity are unclear after the demo, versions and logs are hard to compare,
|
|
151
|
-
tool fleets become noisy and expensive, operators lack visibility into what is
|
|
152
|
-
happening, and raw logs are not a useful analysis surface. Examples:
|
|
153
|
-
|
|
154
|
-
- [ChatGPT + MCP gets painful after the demo](https://www.reddit.com/r/ChatGPTPro/comments/1uz6tzs/where_chatgpt_mcp_gets_painful_after_the_demo/) — auth, versions, logs, and safe tools.
|
|
155
|
-
- [MCP logging and correlation IDs](https://www.reddit.com/r/softwarearchitecture/comments/1r2wnfd/is_mcp_effectively_introducing_a_probabilistic/o51k4zq/) — preserve intent, tool arguments, results, and correlation IDs.
|
|
156
|
-
- [180 tools becomes a permission/context/debugging problem](https://www.reddit.com/r/ClaudeAI/comments/1tuqqpn/i_ship_ai_agents_in_production_the_mess_is_mcp/opbh78y/).
|
|
157
|
-
- [MCP security](https://www.reddit.com/r/cybersecurity/comments/1tgs4gg/mcp_security/) — identity, access, credentials, approved versions, and audit logs.
|
|
158
|
-
- [Raw logs are noisy and hard to query](https://www.reddit.com/r/ChatGPTPro/comments/1ur2tx4/i_made_my_codex_usage_tracker_more_agentnative/).
|
|
159
|
-
|
|
160
|
-
This update adds five focused query surfaces rather than pretending a
|
|
161
|
-
Dashboard solves those problems: `failure_report`, `tool_stats`,
|
|
162
|
-
`security_audit`, `recent_activity`, and `compare_runs`. Trace discovery is also
|
|
163
|
-
recursive, and query results redact obvious credential patterns.
|
|
164
|
-
|
|
165
154
|
## Honest limitations
|
|
166
155
|
|
|
167
156
|
- stdio transport only (Streamable HTTP plus authenticated remote access is the next transport boundary)
|
|
@@ -4,11 +4,17 @@
|
|
|
4
4
|
|
|
5
5
|
[](https://github.com/abhishekash/agenthawk/actions/workflows/ci.yml) [](LICENSE)
|
|
6
6
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
7
|
+
AgentHawk is a small, local-first stdio MCP server for querying JSONL
|
|
8
|
+
OpenTelemetry spans emitted by agent runs. It exposes focused queries for
|
|
9
|
+
run summaries, tool failures, approvals, activity, and comparisons instead of
|
|
10
|
+
requiring an application-specific dashboard.
|
|
11
|
+
|
|
12
|
+
It pairs with [agent-harness](https://github.com/abhishekash/agent-harness),
|
|
13
|
+
but the reader is format-simple: any JSONL of OTel-shaped spans works. The
|
|
14
|
+
project's concrete origin was a shell-boundary regression: the harness blocked
|
|
15
|
+
`cat ../outside.txt`, but the old trace recorded only that `run_shell` was
|
|
16
|
+
called, not the returned diagnostic. AgentHawk and the harness now preserve and
|
|
17
|
+
query that bounded error.
|
|
12
18
|
|
|
13
19
|
## AI-native use cases
|
|
14
20
|
|
|
@@ -36,7 +42,7 @@ cd agenthawk && uv pip install -e .
|
|
|
36
42
|
agenthawk --trace-dir ./traces
|
|
37
43
|
```
|
|
38
44
|
|
|
39
|
-
The `agenthawk` 0.2.
|
|
45
|
+
The `agenthawk` 0.2.1 release is the renamed successor to the published
|
|
40
46
|
[`abhishekash-mcp-trace` 0.1.1](https://pypi.org/project/abhishekash-mcp-trace/0.1.1/)
|
|
41
47
|
distribution. Existing clients can continue using the legacy command while
|
|
42
48
|
migrating.
|
|
@@ -119,27 +125,10 @@ traces/*.jsonl ──▶ agenthawk.core (pure query functions, zero deps)
|
|
|
119
125
|
- **core/server split**: all logic is pure functions over parsed spans; the MCP layer only parses args and JSON-encodes results. Tests hit both layers.
|
|
120
126
|
- **trace_id prefixes**: agents fumble full 32-char hex ids; every tool accepts prefixes.
|
|
121
127
|
- **bounded output**: diagnostics are truncated and obvious credentials are redacted before query results leave the server.
|
|
128
|
+
- **honest cost totals**: `token_usage` returns `cost_known: false` and a null cost when any producer marks pricing as unavailable; unknown pricing is never presented as `$0`.
|
|
122
129
|
- **cursor polling**: `recent_activity` makes the snapshot reader useful while a run is still writing spans.
|
|
123
130
|
- The demo fixture ([`examples/example_trace.jsonl`](examples/example_trace.jsonl)) is a *real* agent-harness run, not hand-written.
|
|
124
131
|
|
|
125
|
-
## Research-driven gaps addressed
|
|
126
|
-
|
|
127
|
-
A small Reddit review surfaced the same production problems repeatedly: auth and
|
|
128
|
-
identity are unclear after the demo, versions and logs are hard to compare,
|
|
129
|
-
tool fleets become noisy and expensive, operators lack visibility into what is
|
|
130
|
-
happening, and raw logs are not a useful analysis surface. Examples:
|
|
131
|
-
|
|
132
|
-
- [ChatGPT + MCP gets painful after the demo](https://www.reddit.com/r/ChatGPTPro/comments/1uz6tzs/where_chatgpt_mcp_gets_painful_after_the_demo/) — auth, versions, logs, and safe tools.
|
|
133
|
-
- [MCP logging and correlation IDs](https://www.reddit.com/r/softwarearchitecture/comments/1r2wnfd/is_mcp_effectively_introducing_a_probabilistic/o51k4zq/) — preserve intent, tool arguments, results, and correlation IDs.
|
|
134
|
-
- [180 tools becomes a permission/context/debugging problem](https://www.reddit.com/r/ClaudeAI/comments/1tuqqpn/i_ship_ai_agents_in_production_the_mess_is_mcp/opbh78y/).
|
|
135
|
-
- [MCP security](https://www.reddit.com/r/cybersecurity/comments/1tgs4gg/mcp_security/) — identity, access, credentials, approved versions, and audit logs.
|
|
136
|
-
- [Raw logs are noisy and hard to query](https://www.reddit.com/r/ChatGPTPro/comments/1ur2tx4/i_made_my_codex_usage_tracker_more_agentnative/).
|
|
137
|
-
|
|
138
|
-
This update adds five focused query surfaces rather than pretending a
|
|
139
|
-
Dashboard solves those problems: `failure_report`, `tool_stats`,
|
|
140
|
-
`security_audit`, `recent_activity`, and `compare_runs`. Trace discovery is also
|
|
141
|
-
recursive, and query results redact obvious credential patterns.
|
|
142
|
-
|
|
143
132
|
## Honest limitations
|
|
144
133
|
|
|
145
134
|
- stdio transport only (Streamable HTTP plus authenticated remote access is the next transport boundary)
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "agenthawk"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "Agent observability MCP server for querying OpenTelemetry traces, failures, tool health, security evidence, and regressions."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -7,13 +7,13 @@
|
|
|
7
7
|
"url": "https://github.com/abhishekash/agenthawk",
|
|
8
8
|
"source": "github"
|
|
9
9
|
},
|
|
10
|
-
"version": "0.2.
|
|
10
|
+
"version": "0.2.1",
|
|
11
11
|
"packages": [
|
|
12
12
|
{
|
|
13
13
|
"registryType": "pypi",
|
|
14
14
|
"registryBaseUrl": "https://pypi.org",
|
|
15
15
|
"identifier": "agenthawk",
|
|
16
|
-
"version": "0.2.
|
|
16
|
+
"version": "0.2.1",
|
|
17
17
|
"runtimeHint": "uvx",
|
|
18
18
|
"transport": {
|
|
19
19
|
"type": "stdio"
|
|
@@ -263,6 +263,7 @@ def token_usage(spans: list[dict[str, Any]]) -> dict[str, Any]:
|
|
|
263
263
|
"""Aggregate token + cost accounting across llm.complete spans."""
|
|
264
264
|
in_tok = out_tok = 0
|
|
265
265
|
cost = 0.0
|
|
266
|
+
cost_known = True
|
|
266
267
|
calls = 0
|
|
267
268
|
for s in spans:
|
|
268
269
|
if s["name"] != "llm.complete":
|
|
@@ -270,13 +271,17 @@ def token_usage(spans: list[dict[str, Any]]) -> dict[str, Any]:
|
|
|
270
271
|
calls += 1
|
|
271
272
|
in_tok += int(s["attributes"].get("llm.usage.input_tokens", 0))
|
|
272
273
|
out_tok += int(s["attributes"].get("llm.usage.output_tokens", 0))
|
|
273
|
-
|
|
274
|
+
attrs = s["attributes"]
|
|
275
|
+
cost += float(attrs.get("llm.cost_usd", 0.0))
|
|
276
|
+
if attrs.get("llm.cost_known", True) is False:
|
|
277
|
+
cost_known = False
|
|
274
278
|
return {
|
|
275
279
|
"llm_calls": calls,
|
|
276
280
|
"input_tokens": in_tok,
|
|
277
281
|
"output_tokens": out_tok,
|
|
278
282
|
"total_tokens": in_tok + out_tok,
|
|
279
|
-
"cost_usd": round(cost, 8),
|
|
283
|
+
"cost_usd": round(cost, 8) if cost_known else None,
|
|
284
|
+
"cost_known": cost_known,
|
|
280
285
|
}
|
|
281
286
|
|
|
282
287
|
|
|
@@ -104,6 +104,15 @@ class TestTokenUsage:
|
|
|
104
104
|
assert usage["llm_calls"] == 5
|
|
105
105
|
assert usage["input_tokens"] == 92 + 94 + 125 + 132 + 196
|
|
106
106
|
assert usage["output_tokens"] > 0
|
|
107
|
+
assert usage["cost_known"] is True
|
|
108
|
+
|
|
109
|
+
def test_unknown_cost_is_not_reported_as_zero(self, spans):
|
|
110
|
+
unknown = dict(spans[0])
|
|
111
|
+
unknown["attributes"] = dict(unknown["attributes"])
|
|
112
|
+
unknown["attributes"]["llm.cost_known"] = False
|
|
113
|
+
usage = core.token_usage([unknown])
|
|
114
|
+
assert usage["cost_known"] is False
|
|
115
|
+
assert usage["cost_usd"] is None
|
|
107
116
|
|
|
108
117
|
|
|
109
118
|
class TestNewQueries:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|