hexgate 0.2.8__tar.gz → 0.2.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hexgate-0.2.8 → hexgate-0.2.9}/PKG-INFO +2 -2
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/google/runner.py +7 -3
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/google/tools.py +1 -1
- hexgate-0.2.9/hexgate/adapters/google/usage.py +44 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/google/wrapper.py +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/langchain/agent.py +9 -3
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/langchain/tools.py +1 -1
- hexgate-0.2.9/hexgate/adapters/langchain/usage.py +84 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/langchain/wrapper.py +1 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/openai/runner.py +71 -4
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/openai/tools.py +1 -1
- hexgate-0.2.9/hexgate/adapters/openai/usage.py +39 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/openai/wrapper.py +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/pydantic_ai/agent.py +20 -2
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/pydantic_ai/tools.py +1 -1
- hexgate-0.2.9/hexgate/adapters/pydantic_ai/usage.py +38 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/pydantic_ai/wrapper.py +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/agents/approvals.py +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/agents/factory.py +27 -5
- hexgate-0.2.9/hexgate/approvals.py +24 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/chat.py +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/pydantic_ai.py +2 -2
- hexgate-0.2.9/hexgate/tools/__init__.py +42 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/decorators.py +10 -3
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tracing/langfuse.py +19 -25
- hexgate-0.2.9/hexgate/tracing/langfuse_core.py +44 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tracing/usage.py +41 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/PKG-INFO +2 -2
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/SOURCES.txt +6 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/requires.txt +1 -1
- {hexgate-0.2.8 → hexgate-0.2.9}/pyproject.toml +2 -2
- hexgate-0.2.8/hexgate/tools/__init__.py +0 -23
- {hexgate-0.2.8 → hexgate-0.2.9}/LICENSE +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/README.md +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/google/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/google/mcp.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/langchain/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/langchain/mcp.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/openai/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/openai/mcp.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/pydantic_ai/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/adapters/pydantic_ai/mcp.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/agents/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/agents/loader.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/agents/models.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/audit.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/bootstrap.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/_common.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/policy/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/policy/main.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/register/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/register/main.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/register/register.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/serve.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cli/state.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cloud/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cloud/attenuate.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cloud/biscuit.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/cloud/client.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/config/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/config/env.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/config/settings.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/builder.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/google.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/langchain.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/models.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/native.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/manifest/openai.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/mcp/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/mcp/client.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/mcp/config.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/mcp/proxy.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/command_policy.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/context.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/sandbox_runtime.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/srt.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/runtime/workspace.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/bans.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/binding.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/builder.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/bundle.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/constraints.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/decision.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/enforcer.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/errors.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/file_scope.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/models.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/policy.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/policy_set.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/rego.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/rego_wasm.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/signing.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/source.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/testing.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/security/wasm_engine.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/streaming/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/streaming/events.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/streaming/normalize.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/_relocated.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/bash.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/_common.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/edit_file.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/glob.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/grep.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/read_file.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tools/files/write_file.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tracing/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/tracing/_senders.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/utils/__init__.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate/utils/retry.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/dependency_links.txt +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/entry_points.txt +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/hexgate.egg-info/top_level.txt +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/setup.cfg +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/tests/test_bootstrap.py +0 -0
- {hexgate-0.2.8 → hexgate-0.2.9}/tests/test_demo.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: hexgate
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.9
|
|
4
4
|
Summary: Hexgate — authorization infrastructure for AI agents (agent runtime + cloud client).
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
Requires-Python: >=3.13
|
|
@@ -23,7 +23,7 @@ Requires-Dist: openai-agents>=0.0.10
|
|
|
23
23
|
Requires-Dist: langgraph>=0.2
|
|
24
24
|
Requires-Dist: nest_asyncio>=1.6
|
|
25
25
|
Requires-Dist: openinference-instrumentation-openai-agents>=0.1
|
|
26
|
-
Requires-Dist: google-adk>=1.
|
|
26
|
+
Requires-Dist: google-adk>=1.14
|
|
27
27
|
Requires-Dist: google-genai>=1.0
|
|
28
28
|
Requires-Dist: litellm>=1.50
|
|
29
29
|
Requires-Dist: openinference-instrumentation-google-adk>=0.1.11
|
|
@@ -9,14 +9,16 @@ from typing import Any, AsyncGenerator, Generator
|
|
|
9
9
|
|
|
10
10
|
import nest_asyncio
|
|
11
11
|
from google.adk.agents import BaseAgent
|
|
12
|
+
from google.adk.apps import App
|
|
12
13
|
from google.adk.runners import Runner
|
|
13
14
|
from google.adk.sessions import BaseSessionService
|
|
14
15
|
from google.genai import types
|
|
15
16
|
from langfuse import get_client, propagate_attributes
|
|
16
17
|
from openinference.instrumentation.google_adk import GoogleADKInstrumentor
|
|
17
18
|
|
|
19
|
+
from hexgate.adapters.google.usage import HexgateUsagePlugin
|
|
18
20
|
from hexgate.adapters.google.wrapper import wrap_google_agent
|
|
19
|
-
from hexgate.
|
|
21
|
+
from hexgate.approvals import ApprovalHandler
|
|
20
22
|
from hexgate.cloud.client import HexgateClient, HexgateConfig
|
|
21
23
|
from hexgate.config.env import resolve_api_key
|
|
22
24
|
from hexgate.runtime import User
|
|
@@ -51,9 +53,11 @@ class HexgateRunner:
|
|
|
51
53
|
approval_handler=approval_handler,
|
|
52
54
|
client=client,
|
|
53
55
|
)
|
|
56
|
+
plugins = list(runner_kwargs.pop("plugins", None) or [])
|
|
57
|
+
plugins.append(HexgateUsagePlugin(api_key=self.api_key))
|
|
58
|
+
app = App(name=app_name, root_agent=self._wrapped_agent, plugins=plugins)
|
|
54
59
|
self._runner = Runner(
|
|
55
|
-
|
|
56
|
-
app_name=app_name,
|
|
60
|
+
app=app,
|
|
57
61
|
session_service=session_service,
|
|
58
62
|
**runner_kwargs,
|
|
59
63
|
)
|
|
@@ -20,7 +20,7 @@ from google.adk.tools.function_tool import FunctionTool
|
|
|
20
20
|
from google.adk.tools.tool_context import ToolContext
|
|
21
21
|
|
|
22
22
|
from hexgate.agents.approvals import resolve_approval_async
|
|
23
|
-
from hexgate.
|
|
23
|
+
from hexgate.approvals import ApprovalHandler
|
|
24
24
|
from hexgate.security.decision import DecisionOutcome
|
|
25
25
|
from hexgate.security.enforcer import PolicyEnforcer
|
|
26
26
|
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Google ADK per-call token usage capture via a ``BasePlugin``.
|
|
2
|
+
|
|
3
|
+
``after_model_callback`` fires once per underlying model call, so a single
|
|
4
|
+
run with several turns (tool-calling loops, sub-agent handoffs) can emit
|
|
5
|
+
more than one usage event. ``callback_context.agent_name`` is read per-call
|
|
6
|
+
rather than fixed at construction, since one ``Runner`` can drive several
|
|
7
|
+
named sub-agents.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from google.adk.agents.callback_context import CallbackContext
|
|
13
|
+
from google.adk.models.llm_response import LlmResponse
|
|
14
|
+
from google.adk.plugins.base_plugin import BasePlugin
|
|
15
|
+
|
|
16
|
+
from hexgate.tracing.usage import emit_llm_usage
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class HexgateUsagePlugin(BasePlugin):
|
|
20
|
+
"""Emits one :class:`~hexgate.tracing.usage.LlmUsageEvent` per
|
|
21
|
+
``after_model_callback`` callback. Never rewrites the response — always
|
|
22
|
+
returns ``None`` so the real model output reaches the agent unchanged."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, *, api_key: str) -> None:
|
|
25
|
+
super().__init__(name="hexgate_usage")
|
|
26
|
+
self._api_key = api_key
|
|
27
|
+
|
|
28
|
+
async def after_model_callback(
|
|
29
|
+
self,
|
|
30
|
+
*,
|
|
31
|
+
callback_context: CallbackContext,
|
|
32
|
+
llm_response: LlmResponse,
|
|
33
|
+
) -> LlmResponse | None:
|
|
34
|
+
usage = llm_response.usage_metadata
|
|
35
|
+
if usage is None:
|
|
36
|
+
return None
|
|
37
|
+
emit_llm_usage(
|
|
38
|
+
callback_context.agent_name,
|
|
39
|
+
llm_response.model_version or "",
|
|
40
|
+
usage.prompt_token_count or 0,
|
|
41
|
+
usage.candidates_token_count or 0,
|
|
42
|
+
api_key=self._api_key,
|
|
43
|
+
)
|
|
44
|
+
return None
|
|
@@ -15,7 +15,7 @@ from typing import TYPE_CHECKING
|
|
|
15
15
|
from google.adk.agents import BaseAgent
|
|
16
16
|
|
|
17
17
|
from hexgate.adapters.google.tools import wrap_tools
|
|
18
|
-
from hexgate.
|
|
18
|
+
from hexgate.approvals import ApprovalHandler
|
|
19
19
|
from hexgate.security.binding import PolicyBinding, resolve_policy
|
|
20
20
|
from hexgate.security.enforcer import build_enforcer
|
|
21
21
|
|
|
@@ -9,6 +9,7 @@ from langfuse import get_client, propagate_attributes
|
|
|
9
9
|
from langfuse.langchain import CallbackHandler
|
|
10
10
|
from langgraph.graph.state import CompiledStateGraph
|
|
11
11
|
|
|
12
|
+
from hexgate.adapters.langchain.usage import HexgateUsageCallbackHandler
|
|
12
13
|
from hexgate.runtime import User
|
|
13
14
|
|
|
14
15
|
if TYPE_CHECKING:
|
|
@@ -33,6 +34,7 @@ class HexgateLangchainAgent:
|
|
|
33
34
|
agent: CompiledStateGraph,
|
|
34
35
|
api_key: str,
|
|
35
36
|
tool_names: list[str],
|
|
37
|
+
agent_name: str = "default",
|
|
36
38
|
binding: PolicyBinding | None = None,
|
|
37
39
|
ban_gate: BanGate | None = None,
|
|
38
40
|
) -> None:
|
|
@@ -43,6 +45,9 @@ class HexgateLangchainAgent:
|
|
|
43
45
|
self._tool_names = tool_names
|
|
44
46
|
self._langfuse = get_client()
|
|
45
47
|
self._callback_handler = CallbackHandler()
|
|
48
|
+
self._usage_handler = HexgateUsageCallbackHandler(
|
|
49
|
+
agent_name=agent_name, api_key=api_key
|
|
50
|
+
)
|
|
46
51
|
|
|
47
52
|
async def _refresh_async(self) -> None:
|
|
48
53
|
"""Refresh the policy binding, if attached (async entry points)."""
|
|
@@ -72,11 +77,12 @@ class HexgateLangchainAgent:
|
|
|
72
77
|
}
|
|
73
78
|
|
|
74
79
|
def _with_callbacks(self, config: RunnableConfig | None) -> RunnableConfig:
|
|
75
|
-
"""Append the Hexgate callback
|
|
80
|
+
"""Append the Hexgate callback handlers to ``config['callbacks']``."""
|
|
76
81
|
merged: RunnableConfig = dict(config) if config else {}
|
|
77
82
|
callbacks = list(merged.get("callbacks") or [])
|
|
78
|
-
|
|
79
|
-
callbacks
|
|
83
|
+
for handler in (self._callback_handler, self._usage_handler):
|
|
84
|
+
if handler not in callbacks:
|
|
85
|
+
callbacks.append(handler)
|
|
80
86
|
merged["callbacks"] = callbacks
|
|
81
87
|
return merged
|
|
82
88
|
|
|
@@ -26,7 +26,7 @@ from hexgate.agents.approvals import (
|
|
|
26
26
|
from hexgate.agents.approvals import (
|
|
27
27
|
resolve_approval_sync as _resolve_approval_sync,
|
|
28
28
|
)
|
|
29
|
-
from hexgate.
|
|
29
|
+
from hexgate.approvals import ApprovalHandler
|
|
30
30
|
from hexgate.security.decision import DecisionOutcome
|
|
31
31
|
from hexgate.security.enforcer import PolicyEnforcer
|
|
32
32
|
from hexgate.tools.decorators import TOOL_METADATA_ATTR
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""LangChain per-call token usage capture via a ``BaseCallbackHandler``.
|
|
2
|
+
|
|
3
|
+
LangGraph propagates ``config["callbacks"]`` down through every node, so a
|
|
4
|
+
handler placed there (see ``HexgateLangchainAgent._with_callbacks``) has its
|
|
5
|
+
``on_llm_end`` invoked by the framework itself on each underlying chat-model
|
|
6
|
+
call — not something this module drives directly. A single ``.invoke()`` can
|
|
7
|
+
therefore emit more than one usage event, once per LLM turn in the run.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from langchain_core.callbacks import BaseCallbackHandler
|
|
15
|
+
from langchain_core.outputs import LLMResult
|
|
16
|
+
|
|
17
|
+
from hexgate.tracing.usage import emit_llm_usage
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class HexgateUsageCallbackHandler(BaseCallbackHandler):
|
|
21
|
+
"""Emits one :class:`~hexgate.tracing.usage.LlmUsageEvent` per
|
|
22
|
+
``on_llm_end`` callback.
|
|
23
|
+
|
|
24
|
+
Reads the standardized ``UsageMetadata`` off the response message when
|
|
25
|
+
the provider populates it, falling back to ``llm_output["token_usage"]``
|
|
26
|
+
for providers that only fill the legacy field. Does nothing when
|
|
27
|
+
neither is present — a provider that reports no usage must not
|
|
28
|
+
synthesize a zeroed event.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__(self, *, agent_name: str, api_key: str | None = None) -> None:
|
|
32
|
+
self._agent_name = agent_name
|
|
33
|
+
self._api_key = api_key
|
|
34
|
+
|
|
35
|
+
async def on_llm_end(self, response: LLMResult, **kwargs: Any) -> None:
|
|
36
|
+
# Async so LangChain's AsyncCallbackManager awaits this inline on the
|
|
37
|
+
# real event loop (asyncio.iscoroutinefunction check in
|
|
38
|
+
# _ahandle_event_for_handler) instead of dispatching it to a thread
|
|
39
|
+
# pool executor — a plain sync def here runs off-loop, and
|
|
40
|
+
# emit_llm_usage's sender never gets a valid loop to schedule its
|
|
41
|
+
# HTTP send on, silently dropping every event.
|
|
42
|
+
usage = _extract_usage(response)
|
|
43
|
+
if usage is None:
|
|
44
|
+
return
|
|
45
|
+
input_tokens, output_tokens = usage
|
|
46
|
+
model = _model_name(response)
|
|
47
|
+
emit_llm_usage(
|
|
48
|
+
self._agent_name,
|
|
49
|
+
model,
|
|
50
|
+
input_tokens,
|
|
51
|
+
output_tokens,
|
|
52
|
+
api_key=self._api_key,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _model_name(response: LLMResult) -> str:
|
|
57
|
+
"""Read the model name off the response.
|
|
58
|
+
|
|
59
|
+
``llm_output`` is ``None`` for a streamed response (LangChain's default
|
|
60
|
+
``_combine_llm_outputs`` — and even providers that override it, like
|
|
61
|
+
ChatOpenAI, only combine per-chunk outputs that are non-``None``, which
|
|
62
|
+
streaming chunks typically aren't) — confirmed empirically against a
|
|
63
|
+
real streaming ``ChatOpenAI`` call, not assumed. ``response_metadata``
|
|
64
|
+
on the message is populated in both the streaming and non-streaming
|
|
65
|
+
case, so it's the primary source; ``llm_output`` stays as a fallback
|
|
66
|
+
for providers that only populate the legacy field.
|
|
67
|
+
"""
|
|
68
|
+
message = response.generations[0][0].message # type: ignore[union-attr]
|
|
69
|
+
response_metadata = getattr(message, "response_metadata", None) or {}
|
|
70
|
+
return response_metadata.get("model_name") or (response.llm_output or {}).get(
|
|
71
|
+
"model_name", ""
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _extract_usage(response: LLMResult) -> tuple[int, int] | None:
|
|
76
|
+
"""(input_tokens, output_tokens), or None when the provider reported no
|
|
77
|
+
usage at all."""
|
|
78
|
+
usage_metadata = response.generations[0][0].message.usage_metadata # type: ignore[union-attr]
|
|
79
|
+
if usage_metadata:
|
|
80
|
+
return usage_metadata["input_tokens"], usage_metadata["output_tokens"]
|
|
81
|
+
token_usage = (response.llm_output or {}).get("token_usage")
|
|
82
|
+
if not token_usage:
|
|
83
|
+
return None
|
|
84
|
+
return token_usage.get("prompt_tokens", 0), token_usage.get("completion_tokens", 0)
|
|
@@ -59,6 +59,7 @@ def wrap_langchain_agent(
|
|
|
59
59
|
return HexgateLangchainAgent(
|
|
60
60
|
agent=agent,
|
|
61
61
|
api_key=resolved_key,
|
|
62
|
+
agent_name=agent_name,
|
|
62
63
|
tool_names=tool_names,
|
|
63
64
|
binding=PolicyBinding(enforcer, resolved.source),
|
|
64
65
|
ban_gate=resolve_ban_gate(agent_name, api_key=resolved_key, client=client),
|
|
@@ -14,6 +14,7 @@ import nest_asyncio
|
|
|
14
14
|
from agents import (
|
|
15
15
|
Agent,
|
|
16
16
|
RunConfig,
|
|
17
|
+
RunHooks,
|
|
17
18
|
Runner,
|
|
18
19
|
RunResult,
|
|
19
20
|
RunResultStreaming,
|
|
@@ -21,11 +22,13 @@ from agents import (
|
|
|
21
22
|
TContext,
|
|
22
23
|
TResponseInputItem,
|
|
23
24
|
)
|
|
25
|
+
from agents.lifecycle import RunHooksBase
|
|
24
26
|
from langfuse import get_client, propagate_attributes
|
|
25
27
|
from openinference.instrumentation.openai_agents import OpenAIAgentsInstrumentor
|
|
26
28
|
|
|
29
|
+
from hexgate.adapters.openai.usage import HexgateUsageHooks
|
|
27
30
|
from hexgate.adapters.openai.wrapper import wrap_openai_agent
|
|
28
|
-
from hexgate.
|
|
31
|
+
from hexgate.approvals import ApprovalHandler
|
|
29
32
|
from hexgate.cloud.client import HexgateClient, HexgateConfig
|
|
30
33
|
from hexgate.config.env import resolve_api_key
|
|
31
34
|
from hexgate.runtime import User
|
|
@@ -34,6 +37,47 @@ from hexgate.security.binding import PolicyBinding, resolve_policy
|
|
|
34
37
|
from hexgate.security.enforcer import build_enforcer
|
|
35
38
|
|
|
36
39
|
|
|
40
|
+
class _CompositeRunHooks(RunHooks):
|
|
41
|
+
"""Fan a run's lifecycle callbacks out to multiple ``RunHooks``.
|
|
42
|
+
|
|
43
|
+
``Runner.run*`` accepts exactly one ``hooks=`` object; when the caller
|
|
44
|
+
already passed one, ``HexgateUsageHooks`` must not replace it — this
|
|
45
|
+
composes both and forwards every ``RunHooksBase`` callback to each in
|
|
46
|
+
turn.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(self, hooks: list[RunHooksBase]) -> None:
|
|
50
|
+
self._hooks = hooks
|
|
51
|
+
|
|
52
|
+
async def on_llm_start(self, context, agent, system_prompt, input_items) -> None:
|
|
53
|
+
for hook in self._hooks:
|
|
54
|
+
await hook.on_llm_start(context, agent, system_prompt, input_items)
|
|
55
|
+
|
|
56
|
+
async def on_llm_end(self, context, agent, response) -> None:
|
|
57
|
+
for hook in self._hooks:
|
|
58
|
+
await hook.on_llm_end(context, agent, response)
|
|
59
|
+
|
|
60
|
+
async def on_agent_start(self, context, agent) -> None:
|
|
61
|
+
for hook in self._hooks:
|
|
62
|
+
await hook.on_agent_start(context, agent)
|
|
63
|
+
|
|
64
|
+
async def on_agent_end(self, context, agent, output) -> None:
|
|
65
|
+
for hook in self._hooks:
|
|
66
|
+
await hook.on_agent_end(context, agent, output)
|
|
67
|
+
|
|
68
|
+
async def on_handoff(self, context, from_agent, to_agent) -> None:
|
|
69
|
+
for hook in self._hooks:
|
|
70
|
+
await hook.on_handoff(context, from_agent, to_agent)
|
|
71
|
+
|
|
72
|
+
async def on_tool_start(self, context, agent, tool) -> None:
|
|
73
|
+
for hook in self._hooks:
|
|
74
|
+
await hook.on_tool_start(context, agent, tool)
|
|
75
|
+
|
|
76
|
+
async def on_tool_end(self, context, agent, tool, result) -> None:
|
|
77
|
+
for hook in self._hooks:
|
|
78
|
+
await hook.on_tool_end(context, agent, tool, result)
|
|
79
|
+
|
|
80
|
+
|
|
37
81
|
class HexgateRunner:
|
|
38
82
|
"""Runner for OpenAI agents with Hexgate tool policy and observability."""
|
|
39
83
|
|
|
@@ -109,12 +153,21 @@ class HexgateRunner:
|
|
|
109
153
|
with propagate_attributes(**kwargs):
|
|
110
154
|
yield
|
|
111
155
|
|
|
156
|
+
def _merge_hooks(self, hooks: RunHooks | None) -> RunHooks:
|
|
157
|
+
"""Compose caller-supplied ``hooks`` with the usage hook — never
|
|
158
|
+
clobber a hooks object the caller already passed."""
|
|
159
|
+
usage_hooks = HexgateUsageHooks(api_key=self.api_key)
|
|
160
|
+
if hooks is None:
|
|
161
|
+
return usage_hooks
|
|
162
|
+
return _CompositeRunHooks([hooks, usage_hooks])
|
|
163
|
+
|
|
112
164
|
async def run(
|
|
113
165
|
self,
|
|
114
166
|
agent: Agent,
|
|
115
167
|
input: str | list[TResponseInputItem] | RunState[TContext],
|
|
116
168
|
user: User,
|
|
117
169
|
run_config: RunConfig | None = None,
|
|
170
|
+
hooks: RunHooks | None = None,
|
|
118
171
|
**kwargs,
|
|
119
172
|
) -> RunResult:
|
|
120
173
|
"""Run the OpenAI agent asynchronously inside a User scope."""
|
|
@@ -132,7 +185,11 @@ class HexgateRunner:
|
|
|
132
185
|
async with user:
|
|
133
186
|
with self._propagate(user, agent.name):
|
|
134
187
|
return await Runner.run(
|
|
135
|
-
wrapped_agent,
|
|
188
|
+
wrapped_agent,
|
|
189
|
+
input,
|
|
190
|
+
run_config=run_config,
|
|
191
|
+
hooks=self._merge_hooks(hooks),
|
|
192
|
+
**kwargs,
|
|
136
193
|
)
|
|
137
194
|
|
|
138
195
|
def run_sync(
|
|
@@ -141,6 +198,7 @@ class HexgateRunner:
|
|
|
141
198
|
input: str | list[TResponseInputItem] | RunState[TContext],
|
|
142
199
|
user: User,
|
|
143
200
|
run_config: RunConfig | None = None,
|
|
201
|
+
hooks: RunHooks | None = None,
|
|
144
202
|
**kwargs,
|
|
145
203
|
) -> RunResult:
|
|
146
204
|
"""Run the OpenAI agent synchronously inside a User scope."""
|
|
@@ -158,7 +216,11 @@ class HexgateRunner:
|
|
|
158
216
|
with user.sync_scope():
|
|
159
217
|
with self._propagate(user, agent.name):
|
|
160
218
|
return Runner.run_sync(
|
|
161
|
-
wrapped_agent,
|
|
219
|
+
wrapped_agent,
|
|
220
|
+
input,
|
|
221
|
+
run_config=run_config,
|
|
222
|
+
hooks=self._merge_hooks(hooks),
|
|
223
|
+
**kwargs,
|
|
162
224
|
)
|
|
163
225
|
|
|
164
226
|
def run_streamed(
|
|
@@ -167,6 +229,7 @@ class HexgateRunner:
|
|
|
167
229
|
input: str | list[TResponseInputItem] | RunState[TContext],
|
|
168
230
|
user: User,
|
|
169
231
|
run_config: RunConfig | None = None,
|
|
232
|
+
hooks: RunHooks | None = None,
|
|
170
233
|
**kwargs,
|
|
171
234
|
) -> RunResultStreaming:
|
|
172
235
|
"""Stream the OpenAI agent inside a User scope.
|
|
@@ -193,7 +256,11 @@ class HexgateRunner:
|
|
|
193
256
|
with user.sync_scope():
|
|
194
257
|
with self._propagate(user, agent.name):
|
|
195
258
|
result = Runner.run_streamed(
|
|
196
|
-
wrapped_agent,
|
|
259
|
+
wrapped_agent,
|
|
260
|
+
input,
|
|
261
|
+
run_config=run_config,
|
|
262
|
+
hooks=self._merge_hooks(hooks),
|
|
263
|
+
**kwargs,
|
|
197
264
|
)
|
|
198
265
|
|
|
199
266
|
original_stream_events = result.stream_events
|
|
@@ -19,7 +19,7 @@ from agents import FunctionTool
|
|
|
19
19
|
from agents.tool import ToolContext
|
|
20
20
|
|
|
21
21
|
from hexgate.agents.approvals import resolve_approval_async
|
|
22
|
-
from hexgate.
|
|
22
|
+
from hexgate.approvals import ApprovalHandler
|
|
23
23
|
from hexgate.security.decision import DecisionOutcome
|
|
24
24
|
from hexgate.security.enforcer import PolicyEnforcer
|
|
25
25
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""OpenAI Agents SDK per-call token usage capture via ``RunHooks``.
|
|
2
|
+
|
|
3
|
+
``Runner.run``/``run_sync``/``run_streamed`` invoke ``on_llm_end`` once per
|
|
4
|
+
underlying model call, so a single run with several turns (tool-calling
|
|
5
|
+
loops, handoffs) can emit more than one usage event.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from agents import Agent, RunContextWrapper
|
|
11
|
+
from agents.items import ModelResponse
|
|
12
|
+
from agents.lifecycle import RunHooks
|
|
13
|
+
|
|
14
|
+
from hexgate.tracing.usage import emit_llm_usage
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class HexgateUsageHooks(RunHooks):
|
|
18
|
+
"""Emits one :class:`~hexgate.tracing.usage.LlmUsageEvent` per
|
|
19
|
+
``on_llm_end`` callback."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, *, api_key: str) -> None:
|
|
22
|
+
self._api_key = api_key
|
|
23
|
+
|
|
24
|
+
async def on_llm_end(
|
|
25
|
+
self,
|
|
26
|
+
context: RunContextWrapper,
|
|
27
|
+
agent: Agent,
|
|
28
|
+
response: ModelResponse,
|
|
29
|
+
) -> None:
|
|
30
|
+
# agent.model is `str | Model | None` — only the str case gives a
|
|
31
|
+
# clean name; a Model implementation has no guaranteed name field.
|
|
32
|
+
model = agent.model if isinstance(agent.model, str) else ""
|
|
33
|
+
emit_llm_usage(
|
|
34
|
+
agent.name,
|
|
35
|
+
model,
|
|
36
|
+
response.usage.input_tokens,
|
|
37
|
+
response.usage.output_tokens,
|
|
38
|
+
api_key=self._api_key,
|
|
39
|
+
)
|
|
@@ -15,7 +15,7 @@ import dataclasses
|
|
|
15
15
|
from agents import Agent
|
|
16
16
|
|
|
17
17
|
from hexgate.adapters.openai.tools import wrap_tools
|
|
18
|
-
from hexgate.
|
|
18
|
+
from hexgate.approvals import ApprovalHandler
|
|
19
19
|
from hexgate.security.enforcer import PolicyEnforcer
|
|
20
20
|
|
|
21
21
|
|
|
@@ -10,6 +10,7 @@ from pydantic_ai import Agent
|
|
|
10
10
|
from pydantic_ai.agent import AgentRun, AgentRunResult
|
|
11
11
|
from pydantic_ai.result import StreamedRunResult
|
|
12
12
|
|
|
13
|
+
from hexgate.adapters.pydantic_ai.usage import emit_run_usage
|
|
13
14
|
from hexgate.runtime import User
|
|
14
15
|
|
|
15
16
|
if TYPE_CHECKING:
|
|
@@ -100,7 +101,9 @@ class HexgatePydanticAgent:
|
|
|
100
101
|
await self._refresh_async()
|
|
101
102
|
await self._check_ban_async(user)
|
|
102
103
|
async with self._abind(user, "run"):
|
|
103
|
-
|
|
104
|
+
result = await self._agent.run(*args, **kwargs)
|
|
105
|
+
emit_run_usage(self._agent_name, self._agent, result, api_key=self._api_key)
|
|
106
|
+
return result
|
|
104
107
|
|
|
105
108
|
def run_sync(
|
|
106
109
|
self,
|
|
@@ -112,7 +115,9 @@ class HexgatePydanticAgent:
|
|
|
112
115
|
self._refresh()
|
|
113
116
|
self._check_ban(user)
|
|
114
117
|
with self._bind(user, "run_sync"):
|
|
115
|
-
|
|
118
|
+
result = self._agent.run_sync(*args, **kwargs)
|
|
119
|
+
emit_run_usage(self._agent_name, self._agent, result, api_key=self._api_key)
|
|
120
|
+
return result
|
|
116
121
|
|
|
117
122
|
@asynccontextmanager
|
|
118
123
|
async def run_stream(
|
|
@@ -127,6 +132,15 @@ class HexgatePydanticAgent:
|
|
|
127
132
|
async with self._abind(user, "run_stream"):
|
|
128
133
|
async with self._agent.run_stream(*args, **kwargs) as result:
|
|
129
134
|
yield result
|
|
135
|
+
# Emit usage only if the run completed, not if the caller aborted mid-stream,
|
|
136
|
+
# because the usage counts from pydantic's side are 0 until the run completes.
|
|
137
|
+
# This can happen if a user cancels a LLM request mid-response: we will never
|
|
138
|
+
# know the total number of input / output tokens, however they are still charged by the LLM provider.
|
|
139
|
+
# This is a known limitation of pydantic_ai's usage reporting, and we will not be able to report usage in this case.
|
|
140
|
+
if result.is_complete:
|
|
141
|
+
emit_run_usage(
|
|
142
|
+
self._agent_name, self._agent, result, api_key=self._api_key
|
|
143
|
+
)
|
|
130
144
|
|
|
131
145
|
@asynccontextmanager
|
|
132
146
|
async def iter(
|
|
@@ -141,6 +155,10 @@ class HexgatePydanticAgent:
|
|
|
141
155
|
async with self._abind(user, "iter"):
|
|
142
156
|
async with self._agent.iter(*args, **kwargs) as run:
|
|
143
157
|
yield run
|
|
158
|
+
if run.result is not None:
|
|
159
|
+
emit_run_usage(
|
|
160
|
+
self._agent_name, self._agent, run, api_key=self._api_key
|
|
161
|
+
)
|
|
144
162
|
|
|
145
163
|
def __getattr__(self, name: str) -> Any:
|
|
146
164
|
"""Delegate unknown attributes to the wrapped agent.
|
|
@@ -20,7 +20,7 @@ from pydantic_ai.exceptions import ModelRetry
|
|
|
20
20
|
from pydantic_ai.tools import Tool
|
|
21
21
|
|
|
22
22
|
from hexgate.agents.approvals import resolve_approval_async
|
|
23
|
-
from hexgate.
|
|
23
|
+
from hexgate.approvals import ApprovalHandler
|
|
24
24
|
from hexgate.security.decision import DecisionOutcome
|
|
25
25
|
from hexgate.security.enforcer import PolicyEnforcer
|
|
26
26
|
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Pydantic AI has no per-call usage hook — usage is read from the run
|
|
2
|
+
result after the call completes and reported as one aggregate event per
|
|
3
|
+
agent run (not per LLM call), a documented limitation vs. the other three
|
|
4
|
+
adapters.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from pydantic_ai import Agent
|
|
12
|
+
|
|
13
|
+
from hexgate.manifest.pydantic_ai import extract_model
|
|
14
|
+
from hexgate.tracing.usage import emit_llm_usage
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def emit_run_usage(agent_name: str, agent: Agent, result: Any, *, api_key: str) -> None:
|
|
18
|
+
"""Emit one aggregate LlmUsageEvent for a completed pydantic_ai run.
|
|
19
|
+
|
|
20
|
+
``result`` is anything exposing ``.usage()`` and ``.response`` —
|
|
21
|
+
``AgentRunResult``, ``StreamedRunResult``, and ``AgentRun`` (from
|
|
22
|
+
``run``/``run_sync``, ``run_stream``, and ``iter`` respectively) all
|
|
23
|
+
qualify. Model name comes from the actual run's response when
|
|
24
|
+
available (pydantic_ai supports per-call model overrides), else the
|
|
25
|
+
agent's statically configured model.
|
|
26
|
+
"""
|
|
27
|
+
usage = result.usage()
|
|
28
|
+
response = getattr(result, "response", None)
|
|
29
|
+
model_name = (
|
|
30
|
+
getattr(response, "model_name", None) or extract_model(agent.model) or ""
|
|
31
|
+
)
|
|
32
|
+
emit_llm_usage(
|
|
33
|
+
agent_name,
|
|
34
|
+
model_name,
|
|
35
|
+
usage.input_tokens,
|
|
36
|
+
usage.output_tokens,
|
|
37
|
+
api_key=api_key,
|
|
38
|
+
)
|
|
@@ -17,7 +17,7 @@ from pydantic_ai.tools import Tool
|
|
|
17
17
|
|
|
18
18
|
from hexgate.adapters.pydantic_ai.agent import HexgatePydanticAgent
|
|
19
19
|
from hexgate.adapters.pydantic_ai.tools import wrap_tools
|
|
20
|
-
from hexgate.
|
|
20
|
+
from hexgate.approvals import ApprovalHandler
|
|
21
21
|
from hexgate.cloud.client import HexgateClient, HexgateConfig
|
|
22
22
|
from hexgate.config.env import resolve_api_key
|
|
23
23
|
from hexgate.security.bans import resolve_ban_gate
|