agentdynamics 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdynamics/__init__.py +12 -0
- agentdynamics/__main__.py +322 -0
- agentdynamics/analysis.py +755 -0
- agentdynamics/autotrace.py +697 -0
- agentdynamics/bootstrap/sitecustomize.py +27 -0
- agentdynamics/collectors/__init__.py +0 -0
- agentdynamics/collectors/aegis_audit.py +72 -0
- agentdynamics/collectors/claude_code.py +257 -0
- agentdynamics/collectors/generic.py +103 -0
- agentdynamics/collectors/inbox.py +95 -0
- agentdynamics/collectors/langfuse.py +98 -0
- agentdynamics/collectors/langsmith.py +258 -0
- agentdynamics/collectors/otlp.py +349 -0
- agentdynamics/collectors/spans.py +202 -0
- agentdynamics/config.py +139 -0
- agentdynamics/engine.py +376 -0
- agentdynamics/flows.py +142 -0
- agentdynamics/govern.py +231 -0
- agentdynamics/integrations/__init__.py +1 -0
- agentdynamics/integrations/aegis.py +352 -0
- agentdynamics/phases.py +82 -0
- agentdynamics/pricing.py +69 -0
- agentdynamics/privacy.py +41 -0
- agentdynamics/sdk.py +105 -0
- agentdynamics/server.py +949 -0
- agentdynamics/slo.py +84 -0
- agentdynamics/store.py +228 -0
- agentdynamics/web/app.js +1069 -0
- agentdynamics/web/charts.js +185 -0
- agentdynamics/web/index.html +40 -0
- agentdynamics/web/style.css +244 -0
- agentdynamics-0.4.0.dist-info/METADATA +195 -0
- agentdynamics-0.4.0.dist-info/RECORD +37 -0
- agentdynamics-0.4.0.dist-info/WHEEL +5 -0
- agentdynamics-0.4.0.dist-info/entry_points.txt +2 -0
- agentdynamics-0.4.0.dist-info/licenses/LICENSE +202 -0
- agentdynamics-0.4.0.dist-info/top_level.txt +1 -0
agentdynamics/govern.py
ADDED
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Observe -> govern: turn what agents actually did into a least-privilege Aegis policy.
|
|
2
|
+
|
|
3
|
+
Given the policy runs were governed by (the *base*) and the runs themselves, `synthesize` produces a
|
|
4
|
+
policy that can only be tighter than the base:
|
|
5
|
+
|
|
6
|
+
tools only tools that were actually called (and allowed); unused grants are dropped
|
|
7
|
+
args observed path/URL prefixes narrow a base prefix; small categorical sets become `one_of`;
|
|
8
|
+
`max_len` shrinks to 2x the longest observed value; base regexes are kept as they are
|
|
9
|
+
budget p99 of real per-task usage x headroom, never above the base
|
|
10
|
+
spawn depth / fan-out actually used, never above the base; no spawning if none was observed
|
|
11
|
+
data unchanged (egress sinks are kept even for dropped tools; removing one reads as widening)
|
|
12
|
+
|
|
13
|
+
Without a base it drafts a standalone policy from observation alone (review it before use).
|
|
14
|
+
Every change is reported so the diff can be reviewed like any other PR. Verify the result with
|
|
15
|
+
`aegis ratify` and `aegis drift --baseline <base> --candidate <generated>`.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import copy
|
|
20
|
+
import json
|
|
21
|
+
import math
|
|
22
|
+
import os
|
|
23
|
+
from collections import defaultdict
|
|
24
|
+
|
|
25
|
+
from .analysis import pct
|
|
26
|
+
|
|
27
|
+
INTERNAL_TOOLS = {"model.spend", "agent.spawn", "agent.revoke"}
|
|
28
|
+
OUTWARD_HINTS = ("http", "post", "send", "email", "mail", "slack", "webhook", "upload", "write", "publish", "notify")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _round_up(x, step):
|
|
32
|
+
return math.ceil(x / step) * step if x > 0 else step
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _common_prefix(values):
|
|
36
|
+
if not values:
|
|
37
|
+
return ""
|
|
38
|
+
p = os.path.commonprefix(values)
|
|
39
|
+
cut = p.rfind("/")
|
|
40
|
+
return p[:cut + 1] if cut >= 0 else ""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _looks_pathlike(values):
|
|
44
|
+
return all(isinstance(v, str) and ("/" in v) for v in values)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _arg_constraints(name, values, base_c, n_calls, changes, tool):
|
|
48
|
+
"""Tighten one argument's constraint from observed values."""
|
|
49
|
+
c = dict(base_c or {})
|
|
50
|
+
strs = [v if isinstance(v, str) else json.dumps(v) for v in values]
|
|
51
|
+
if not strs:
|
|
52
|
+
return c
|
|
53
|
+
if _looks_pathlike(strs):
|
|
54
|
+
prefix = _common_prefix(strs)
|
|
55
|
+
base_prefix = c.get("prefix")
|
|
56
|
+
if prefix and len(prefix) > 1 and (base_prefix is None or (prefix.startswith(base_prefix) and prefix != base_prefix)):
|
|
57
|
+
if base_prefix is None and "matches" in c:
|
|
58
|
+
pass # a base regex already shapes this argument; don't guess how a prefix composes with it
|
|
59
|
+
else:
|
|
60
|
+
c["prefix"] = prefix
|
|
61
|
+
changes.append(f"{tool}.{name}: prefix {base_prefix!r} -> {prefix!r} (observed in {len(strs)} calls)")
|
|
62
|
+
distinct = sorted(set(strs))
|
|
63
|
+
if (len(distinct) <= 5 and n_calls >= 5 and all(len(v) <= 64 for v in distinct) and not _looks_pathlike(strs)
|
|
64
|
+
and "one_of" not in c and "matches" not in c and "prefix" not in c):
|
|
65
|
+
c["one_of"] = distinct
|
|
66
|
+
changes.append(f"{tool}.{name}: one_of {distinct}")
|
|
67
|
+
longest = max(len(v) for v in strs)
|
|
68
|
+
new_len = max(64, _round_up(longest * 2, 64))
|
|
69
|
+
if c.get("max_len") is None or new_len < c["max_len"]:
|
|
70
|
+
changes.append(f"{tool}.{name}: max_len {c.get('max_len')} -> {new_len} (longest observed {longest})")
|
|
71
|
+
c["max_len"] = new_len
|
|
72
|
+
return c
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def synthesize(base_doc, tasks, steps, headroom=1.5, name=None):
|
|
76
|
+
"""Return (policy_doc, changes, stats). `steps` are tool/span steps of the given tasks."""
|
|
77
|
+
changes = []
|
|
78
|
+
base = copy.deepcopy(base_doc) if base_doc else None
|
|
79
|
+
base_tools = {t["name"]: t for t in (base or {}).get("tools", {}).get("allow", [])}
|
|
80
|
+
calls = defaultdict(list)
|
|
81
|
+
agents_by_tool = defaultdict(set)
|
|
82
|
+
spawn_depth = 0
|
|
83
|
+
fanout = defaultdict(int)
|
|
84
|
+
for s in steps:
|
|
85
|
+
if s.get("kind") == "span" and s.get("rule") == "spawn.granted":
|
|
86
|
+
fanout[(s.get("task_id"), s.get("agent"))] += 1
|
|
87
|
+
continue
|
|
88
|
+
if s.get("kind") != "tool" or s.get("denied") or s.get("name") in INTERNAL_TOOLS:
|
|
89
|
+
continue
|
|
90
|
+
if s.get("grant_depth"):
|
|
91
|
+
spawn_depth = max(spawn_depth, int(s["grant_depth"]))
|
|
92
|
+
try:
|
|
93
|
+
args = json.loads(s["args_json"]) if s.get("args_json") else {}
|
|
94
|
+
except ValueError:
|
|
95
|
+
args = {}
|
|
96
|
+
calls[s["name"]].append(args if isinstance(args, dict) else {})
|
|
97
|
+
agents_by_tool[s["name"]].add((s.get("grant_depth") or 0) > 0)
|
|
98
|
+
|
|
99
|
+
used = sorted(calls)
|
|
100
|
+
if base_tools:
|
|
101
|
+
unused = sorted(set(base_tools) - set(used) - {"agent.spawn"})
|
|
102
|
+
for t in unused:
|
|
103
|
+
changes.append(f"tools: removed unused grant '{t}'")
|
|
104
|
+
outside = sorted(set(used) - set(base_tools))
|
|
105
|
+
used = [t for t in used if t in base_tools] # never grant something the base did not
|
|
106
|
+
for t in outside:
|
|
107
|
+
changes.append(f"tools: '{t}' was called but is not in the base policy; not granted")
|
|
108
|
+
allow = []
|
|
109
|
+
for tool in used:
|
|
110
|
+
entry = copy.deepcopy(base_tools.get(tool, {"name": tool}))
|
|
111
|
+
observed = calls[tool]
|
|
112
|
+
arg_names = sorted({k for a in observed for k in a})
|
|
113
|
+
args = dict(entry.get("args") or {})
|
|
114
|
+
for a in arg_names:
|
|
115
|
+
vals = [o[a] for o in observed if a in o and o[a] is not None]
|
|
116
|
+
args[a] = _arg_constraints(a, vals, args.get(a), len(observed), changes, tool)
|
|
117
|
+
if args:
|
|
118
|
+
entry["args"] = args
|
|
119
|
+
always = sorted(a for a in arg_names if all(a in o for o in observed))
|
|
120
|
+
req = sorted(set(entry.get("require_args") or []) | set(always))
|
|
121
|
+
if req:
|
|
122
|
+
if req != sorted(entry.get("require_args") or []):
|
|
123
|
+
changes.append(f"{tool}: require_args {sorted(entry.get('require_args') or [])} -> {req}")
|
|
124
|
+
entry["require_args"] = req
|
|
125
|
+
allow.append(entry)
|
|
126
|
+
|
|
127
|
+
# budget from real per-task usage
|
|
128
|
+
cost = [(t.get("cost") or 0) + (t.get("subagent_cost") or 0) for t in tasks]
|
|
129
|
+
obs = {
|
|
130
|
+
"usd": max(0.01, round((pct(cost, 0.99) or 0) * headroom, 4)),
|
|
131
|
+
"tokens": int(_round_up((pct([t.get("total_tokens") or 0 for t in tasks], 0.99) or 0) * headroom, 1000)),
|
|
132
|
+
"wall_clock_s": float(_round_up((pct([t.get("wall_s") or 0 for t in tasks], 0.99) or 0) * headroom, 30)),
|
|
133
|
+
"tool_calls": int(max(5, math.ceil((pct([t.get("tool_calls") or 0 for t in tasks], 0.99) or 0) * headroom))),
|
|
134
|
+
}
|
|
135
|
+
bbase = (base or {}).get("budget") or {}
|
|
136
|
+
budget = {}
|
|
137
|
+
for k, v in obs.items():
|
|
138
|
+
b = bbase.get(k)
|
|
139
|
+
budget[k] = min(v, b) if b else v
|
|
140
|
+
if b is not None and budget[k] < b:
|
|
141
|
+
changes.append(f"budget.{k}: {b} -> {budget[k]} (p99 observed x {headroom})")
|
|
142
|
+
|
|
143
|
+
# spawning
|
|
144
|
+
sbase = (base or {}).get("spawn") or {}
|
|
145
|
+
max_fan = max(fanout.values()) if fanout else 0
|
|
146
|
+
if spawn_depth == 0 and not fanout:
|
|
147
|
+
spawn = {"max_depth": 0, "max_fanout": 0, "max_descendants": 0,
|
|
148
|
+
"child_budget_fraction": sbase.get("child_budget_fraction", 0.5), "allow_tools": []}
|
|
149
|
+
if sbase.get("max_depth"):
|
|
150
|
+
changes.append(f"spawn: no sub-agents observed; max_depth {sbase.get('max_depth')} -> 0")
|
|
151
|
+
else:
|
|
152
|
+
spawn = {"max_depth": min(spawn_depth or 1, sbase.get("max_depth", spawn_depth or 1)),
|
|
153
|
+
"max_fanout": min(max(max_fan, 1), sbase.get("max_fanout", max(max_fan, 1))),
|
|
154
|
+
"max_descendants": min(max(sum(fanout.values()), 1), sbase.get("max_descendants", 10 ** 6)),
|
|
155
|
+
"child_budget_fraction": sbase.get("child_budget_fraction", 0.5)}
|
|
156
|
+
child_tools = sorted(t for t, flags in agents_by_tool.items() if True in flags)
|
|
157
|
+
base_at = sbase.get("allow_tools")
|
|
158
|
+
spawn["allow_tools"] = sorted(set(child_tools) & set(base_at)) if base_at is not None else child_tools
|
|
159
|
+
for k in ("max_depth", "max_fanout", "max_descendants"):
|
|
160
|
+
if sbase.get(k) is not None and spawn[k] < sbase[k]:
|
|
161
|
+
changes.append(f"spawn.{k}: {sbase[k]} -> {spawn[k]}")
|
|
162
|
+
if spawn["max_depth"] > 0 or fanout:
|
|
163
|
+
if "agent.spawn" in base_tools or not base_tools:
|
|
164
|
+
allow.append({"name": "agent.spawn"})
|
|
165
|
+
elif "agent.spawn" in base_tools:
|
|
166
|
+
changes.append("tools: removed unused grant 'agent.spawn'")
|
|
167
|
+
|
|
168
|
+
data = copy.deepcopy((base or {}).get("data")) if base else None
|
|
169
|
+
granted = {e["name"] for e in allow}
|
|
170
|
+
if data:
|
|
171
|
+
pass # keep egress sinks as they are: Aegis drift treats dropping a sink as widening, even for a removed tool
|
|
172
|
+
else:
|
|
173
|
+
data = {"max_classification": "internal",
|
|
174
|
+
"egress": {"sinks": sorted(t for t in granted if any(h in t.lower() for h in OUTWARD_HINTS)),
|
|
175
|
+
"max_classification": "public", "block_pii": ["email", "credit_card", "api_key", "private_key"]}}
|
|
176
|
+
doc = {"name": name or ((base or {}).get("name", "observed") + "-observed"),
|
|
177
|
+
"version": int((base or {}).get("version", 0)) + 1}
|
|
178
|
+
if base and base.get("effects"):
|
|
179
|
+
doc["effects"] = base["effects"]
|
|
180
|
+
doc.update({"tools": {"allow": allow}, "budget": budget, "data": data, "spawn": spawn})
|
|
181
|
+
stats = {"tasks": len(tasks), "tool_calls": sum(len(v) for v in calls.values()), "tools_used": len(used),
|
|
182
|
+
"tools_granted_base": len(base_tools) or None, "observed_budget": obs}
|
|
183
|
+
return doc, changes, stats
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
# ---------------------------------------------------------------- minimal YAML emitter (no dependency)
|
|
187
|
+
|
|
188
|
+
def _scalar(v):
|
|
189
|
+
if v is None:
|
|
190
|
+
return "null"
|
|
191
|
+
if isinstance(v, bool):
|
|
192
|
+
return "true" if v else "false"
|
|
193
|
+
if isinstance(v, (int, float)):
|
|
194
|
+
return repr(v)
|
|
195
|
+
return json.dumps(str(v)) # JSON strings are valid YAML double-quoted scalars
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def to_yaml(obj, indent=0):
|
|
199
|
+
pad = " " * indent
|
|
200
|
+
lines = []
|
|
201
|
+
if isinstance(obj, dict):
|
|
202
|
+
for k, v in obj.items():
|
|
203
|
+
if isinstance(v, (dict, list)) and v:
|
|
204
|
+
lines.append(f"{pad}{k}:")
|
|
205
|
+
lines.append(to_yaml(v, indent + 1))
|
|
206
|
+
elif isinstance(v, (dict, list)):
|
|
207
|
+
lines.append(f"{pad}{k}: {'{}' if isinstance(v, dict) else '[]'}")
|
|
208
|
+
else:
|
|
209
|
+
lines.append(f"{pad}{k}: {_scalar(v)}")
|
|
210
|
+
elif isinstance(obj, list):
|
|
211
|
+
for item in obj:
|
|
212
|
+
if isinstance(item, dict) and item:
|
|
213
|
+
inner = to_yaml(item, indent + 1).split("\n")
|
|
214
|
+
lines.append(f"{pad}- {inner[0].strip()}")
|
|
215
|
+
lines.extend(inner[1:])
|
|
216
|
+
elif isinstance(item, list):
|
|
217
|
+
lines.append(f"{pad}- {json.dumps(item)}")
|
|
218
|
+
else:
|
|
219
|
+
lines.append(f"{pad}- {_scalar(item)}")
|
|
220
|
+
return "\n".join(lines)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def render(doc, changes, stats, source=""):
|
|
224
|
+
head = ["# Generated by `agentdynamics policy export` from observed agent behaviour.",
|
|
225
|
+
f"# {source}".rstrip(),
|
|
226
|
+
f"# Based on {stats['tasks']} tasks and {stats['tool_calls']} allowed tool calls.",
|
|
227
|
+
"# Verify before use: aegis ratify --policy <this file>",
|
|
228
|
+
"# aegis drift --baseline <base> --candidate <this file>",
|
|
229
|
+
"#", "# Changes:"]
|
|
230
|
+
head += [f"# - {c}" for c in changes] or ["# (none)"]
|
|
231
|
+
return "\n".join(h for h in head if h != "#") + "\n\n" + to_yaml(doc) + "\n"
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Integrations with other agent tooling."""
|
|
@@ -0,0 +1,352 @@
|
|
|
1
|
+
"""AgentDynamics x Aegis: observe what agents do, and govern what they may do.
|
|
2
|
+
|
|
3
|
+
import agentdynamics
|
|
4
|
+
from agentdynamics.integrations import aegis as governance
|
|
5
|
+
|
|
6
|
+
agentdynamics.init(project="support")
|
|
7
|
+
kernel, root = build_kernel(load_policy("policy.yaml"), registry) # Aegis
|
|
8
|
+
governance.instrument(kernel, root, watchdog=governance.Watchdog(max_repeated_denials=3))
|
|
9
|
+
|
|
10
|
+
One call wires four flows:
|
|
11
|
+
|
|
12
|
+
1. Govern -> observe. Every Aegis decision (allowed tool call, denial with its rule id, spawn,
|
|
13
|
+
revocation, budget reservation) becomes a step in the AgentDynamics task it happened in, and every
|
|
14
|
+
Aegis audit record carries the task's run id, workflow and node (`details.ctx`) so the two logs join.
|
|
15
|
+
2. Budgets cover model spend. Anthropic / OpenAI / `agentdynamics.llm_call` requests reserve their
|
|
16
|
+
estimated cost against the Aegis ledger *before* they are sent and settle the actual cost after.
|
|
17
|
+
An exhausted budget or a revoked grant stops the request; it is never made.
|
|
18
|
+
3. Detect -> enforce. A watchdog evaluates each run as it happens (repeated denials, loops, runaway
|
|
19
|
+
cost, too many model calls) and revokes the grant through Aegis, which disables the whole agent tree.
|
|
20
|
+
4. Observe -> govern. Runs carry the policy name, version and digest, so the console can compare
|
|
21
|
+
policy versions, find unused grants, and generate a tightened policy (`agentdynamics policy export`).
|
|
22
|
+
|
|
23
|
+
Nothing here can loosen Aegis: it only adds correlation data, reserves budget, and revokes.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import contextvars
|
|
28
|
+
import json
|
|
29
|
+
import math
|
|
30
|
+
import threading
|
|
31
|
+
import time
|
|
32
|
+
from collections import Counter
|
|
33
|
+
|
|
34
|
+
from .. import autotrace as at
|
|
35
|
+
from .. import pricing
|
|
36
|
+
|
|
37
|
+
_grant = contextvars.ContextVar("agentdynamics_aegis_grant", default=None)
|
|
38
|
+
_state = {"kernel": None, "root": None, "installed": False}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _require_aegis():
|
|
42
|
+
try:
|
|
43
|
+
import aegis # noqa: F401
|
|
44
|
+
return aegis
|
|
45
|
+
except ImportError as ex: # pragma: no cover
|
|
46
|
+
raise ImportError("pip install aegis-kernel to use the Aegis integration") from ex
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------- policy identity
|
|
50
|
+
|
|
51
|
+
def policy_info(policy):
|
|
52
|
+
"""Name, version, digest and full content of an Aegis policy (for the run payload)."""
|
|
53
|
+
aegis = _require_aegis()
|
|
54
|
+
doc = aegis.dump_policy(policy) if hasattr(aegis, "dump_policy") else {"name": policy.name, "version": policy.version,
|
|
55
|
+
"tools": {"allow": [{"name": n} for n in sorted(policy.tools)]}}
|
|
56
|
+
digest = aegis.policy_digest(policy) if hasattr(aegis, "policy_digest") else "unknown"
|
|
57
|
+
return {"name": policy.name, "version": policy.version, "digest": digest,
|
|
58
|
+
"label": f"{policy.name}@v{policy.version}#{digest[:8]}", "doc": doc}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
# ---------------------------------------------------------------- grant binding
|
|
62
|
+
|
|
63
|
+
class bind:
|
|
64
|
+
"""Make `grant` the one model calls and the watchdog act on (e.g. inside a spawned sub-agent).
|
|
65
|
+
|
|
66
|
+
child = kernel.spawn(root, SpawnRequest("researcher", ...))
|
|
67
|
+
with governance.bind(child):
|
|
68
|
+
...
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
def __init__(self, grant):
|
|
72
|
+
self.grant = grant
|
|
73
|
+
|
|
74
|
+
def __enter__(self):
|
|
75
|
+
self._tok = _grant.set(self.grant)
|
|
76
|
+
return self.grant
|
|
77
|
+
|
|
78
|
+
def __exit__(self, *a):
|
|
79
|
+
_grant.reset(self._tok)
|
|
80
|
+
return False
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def current_grant():
|
|
84
|
+
return _grant.get() or _state["root"]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ---------------------------------------------------------------- 1. decisions -> steps
|
|
88
|
+
|
|
89
|
+
def _step_from_call(grant, tool, args, t0, result=None, exc=None):
|
|
90
|
+
step = {"kind": "tool", "name": tool, "ts": t0, "end_ts": time.time(), "agent": grant.agent_name,
|
|
91
|
+
"grant_depth": grant.depth, "governed": True,
|
|
92
|
+
"input": args if at._cfg["content"] else {}}
|
|
93
|
+
v = getattr(exc, "verdict", None)
|
|
94
|
+
if v is not None and not v.allowed:
|
|
95
|
+
step.update(denied=True, rule=v.rule, guard=v.guard, error=f"[{v.rule}] {v.reason}"[:300])
|
|
96
|
+
elif exc is not None:
|
|
97
|
+
step.update(is_error=True, error=f"{type(exc).__name__}: {exc}"[:300], rule="kernel.admitted")
|
|
98
|
+
else:
|
|
99
|
+
step.update(rule="kernel.admitted", output_chars=len(at._clip(result, 10 ** 7)), text=at._clip(result, 300))
|
|
100
|
+
return step
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _emit_step(step, fallback_name):
|
|
104
|
+
run = at._current.get()
|
|
105
|
+
if run is not None:
|
|
106
|
+
run.add(step)
|
|
107
|
+
else: # a governed call outside any @trace still becomes a (tiny) task
|
|
108
|
+
r = at._Run(fallback_name)
|
|
109
|
+
r.add(step)
|
|
110
|
+
at._emit(r.payload())
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _wrap_kernel(kernel):
|
|
114
|
+
if getattr(kernel, "_agentdynamics_wrapped", False):
|
|
115
|
+
return
|
|
116
|
+
orig_invoke, orig_ainvoke, orig_spawn, orig_revoke = kernel.invoke, kernel.ainvoke, kernel.spawn, kernel.revoke
|
|
117
|
+
|
|
118
|
+
def invoke(grant, tool, /, **args):
|
|
119
|
+
t0 = time.time()
|
|
120
|
+
with bind(grant):
|
|
121
|
+
try:
|
|
122
|
+
r = orig_invoke(grant, tool, **args)
|
|
123
|
+
except Exception as ex:
|
|
124
|
+
_emit_step(_step_from_call(grant, tool, args, t0, exc=ex), f"aegis.{tool}")
|
|
125
|
+
raise
|
|
126
|
+
_emit_step(_step_from_call(grant, tool, args, t0, result=r), f"aegis.{tool}")
|
|
127
|
+
return r
|
|
128
|
+
|
|
129
|
+
async def ainvoke(grant, tool, /, **args):
|
|
130
|
+
t0 = time.time()
|
|
131
|
+
tok = _grant.set(grant)
|
|
132
|
+
try:
|
|
133
|
+
r = await orig_ainvoke(grant, tool, **args)
|
|
134
|
+
except Exception as ex:
|
|
135
|
+
_emit_step(_step_from_call(grant, tool, args, t0, exc=ex), f"aegis.{tool}")
|
|
136
|
+
raise
|
|
137
|
+
finally:
|
|
138
|
+
_grant.reset(tok)
|
|
139
|
+
_emit_step(_step_from_call(grant, tool, args, t0, result=r), f"aegis.{tool}")
|
|
140
|
+
return r
|
|
141
|
+
|
|
142
|
+
def spawn(parent, req):
|
|
143
|
+
t0 = time.time()
|
|
144
|
+
try:
|
|
145
|
+
child = orig_spawn(parent, req)
|
|
146
|
+
except Exception as ex:
|
|
147
|
+
_emit_step(_step_from_call(parent, "agent.spawn", {"name": req.name, "tools": sorted(req.tools)}, t0, exc=ex),
|
|
148
|
+
"aegis.spawn")
|
|
149
|
+
raise
|
|
150
|
+
_emit_step({"kind": "span", "span_kind": "agent", "name": child.agent_name, "node": None, "ts": t0,
|
|
151
|
+
"start_ts": t0, "end_ts": time.time(), "agent": parent.agent_name, "governed": True,
|
|
152
|
+
"rule": "spawn.granted", "text": f"spawned {child.agent_name} (depth {child.depth}) "
|
|
153
|
+
f"with {sorted(req.tools)}"}, "aegis.spawn")
|
|
154
|
+
return child
|
|
155
|
+
|
|
156
|
+
def revoke(grant, reason="operator"):
|
|
157
|
+
orig_revoke(grant, reason)
|
|
158
|
+
_emit_step({"kind": "notice", "name": "revoked", "ts": time.time(), "agent": grant.agent_name,
|
|
159
|
+
"rule": "grant.revoked_subtree", "text": f"{grant.agent_name}: {reason}"}, "aegis.revoke")
|
|
160
|
+
|
|
161
|
+
kernel.invoke, kernel.ainvoke, kernel.spawn, kernel.revoke = invoke, ainvoke, spawn, revoke
|
|
162
|
+
kernel._agentdynamics_wrapped = True
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _context():
|
|
166
|
+
"""Correlation ids stamped into every Aegis audit record (details.ctx)."""
|
|
167
|
+
run = at._current.get()
|
|
168
|
+
if run is None:
|
|
169
|
+
return None
|
|
170
|
+
return {"run_id": run.id, "workflow": run.name, "node": at._node.get(), "project": at._cfg["project"],
|
|
171
|
+
"source": "agentdynamics"}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# ---------------------------------------------------------------- 2. budgets cover model spend
|
|
175
|
+
|
|
176
|
+
class ModelSpendGate:
|
|
177
|
+
"""Reserve a model call's estimated cost in Aegis before it is sent; settle the real cost after."""
|
|
178
|
+
|
|
179
|
+
def __init__(self, kernel, default_output_tokens=1024):
|
|
180
|
+
self.kernel = kernel
|
|
181
|
+
self.default_output_tokens = default_output_tokens
|
|
182
|
+
|
|
183
|
+
def estimate(self, model, kwargs):
|
|
184
|
+
body = kwargs.get("messages") if kwargs.get("messages") is not None else kwargs.get("input")
|
|
185
|
+
try:
|
|
186
|
+
chars = len(json.dumps(body, default=str)) + len(json.dumps(kwargs.get("system") or "", default=str))
|
|
187
|
+
except (TypeError, ValueError):
|
|
188
|
+
chars = 4000
|
|
189
|
+
tin = math.ceil(chars / 4)
|
|
190
|
+
tout = kwargs.get("max_tokens") or kwargs.get("max_output_tokens") or kwargs.get("max_completion_tokens") \
|
|
191
|
+
or self.default_output_tokens
|
|
192
|
+
return pricing.cost(model or "", tin, tout), tin + tout
|
|
193
|
+
|
|
194
|
+
def before(self, provider, model, kwargs):
|
|
195
|
+
grant = current_grant()
|
|
196
|
+
if grant is None:
|
|
197
|
+
return None
|
|
198
|
+
usd, tokens = self.estimate(model, kwargs)
|
|
199
|
+
return self.kernel.reserve_spend(grant, usd=usd, tokens=tokens, label=str(model))
|
|
200
|
+
|
|
201
|
+
def after(self, reservation, usd, tokens, err):
|
|
202
|
+
if reservation is not None:
|
|
203
|
+
self.kernel.settle_spend(reservation, usd=usd if err is None else 0.0, tokens=tokens if err is None else 0)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
# ---------------------------------------------------------------- 3. detect -> enforce
|
|
207
|
+
|
|
208
|
+
class Watchdog:
|
|
209
|
+
"""Live limits evaluated on every step of a run. When one trips, the grant is revoked through Aegis
|
|
210
|
+
(its whole sub-tree stops) and the reason is recorded in both tools.
|
|
211
|
+
|
|
212
|
+
max_repeated_denials the same tool refused N times in a row (whatever the rule): the agent is probing
|
|
213
|
+
a boundary (a classic prompt-injection symptom) or stuck
|
|
214
|
+
max_denials total denials in one run
|
|
215
|
+
max_node_visits one node/stage executed N times: a loop
|
|
216
|
+
max_run_cost_usd model spend of the run (what AgentDynamics prices, not just the Aegis estimate)
|
|
217
|
+
max_llm_calls model calls in one run
|
|
218
|
+
"""
|
|
219
|
+
|
|
220
|
+
def __init__(self, max_repeated_denials=3, max_denials=None, max_node_visits=None, max_run_cost_usd=None,
|
|
221
|
+
max_llm_calls=None, action="revoke"):
|
|
222
|
+
self.limits = {"max_repeated_denials": max_repeated_denials, "max_denials": max_denials,
|
|
223
|
+
"max_node_visits": max_node_visits, "max_run_cost_usd": max_run_cost_usd,
|
|
224
|
+
"max_llm_calls": max_llm_calls}
|
|
225
|
+
self.action = action
|
|
226
|
+
self.kernel = None
|
|
227
|
+
self.trips = []
|
|
228
|
+
self._lock = threading.Lock()
|
|
229
|
+
|
|
230
|
+
def _stats(self, run):
|
|
231
|
+
st = getattr(run, "_ad_watch", None)
|
|
232
|
+
if st is None:
|
|
233
|
+
st = run._ad_watch = {"streak": 0, "last": None, "denials": 0, "nodes": Counter(), "cost": 0.0,
|
|
234
|
+
"llm": 0, "tripped": False}
|
|
235
|
+
return st
|
|
236
|
+
|
|
237
|
+
def __call__(self, run, step):
|
|
238
|
+
if step.get("kind") == "notice":
|
|
239
|
+
return
|
|
240
|
+
with self._lock:
|
|
241
|
+
st = self._stats(run)
|
|
242
|
+
if st["tripped"]:
|
|
243
|
+
return
|
|
244
|
+
if step.get("denied"):
|
|
245
|
+
# keyed on the tool, not the rule: an agent that varies its payload (/etc/passwd, then
|
|
246
|
+
# /workspace/../etc/passwd) trips different rules but is still probing the same boundary
|
|
247
|
+
key = step.get("name")
|
|
248
|
+
st["streak"] = st["streak"] + 1 if key == st["last"] else 1
|
|
249
|
+
st["last"] = key
|
|
250
|
+
st["denials"] += 1
|
|
251
|
+
elif step.get("kind") == "tool":
|
|
252
|
+
st["streak"], st["last"] = 0, None
|
|
253
|
+
if step.get("kind") == "span" and step.get("node"):
|
|
254
|
+
st["nodes"][step["node"]] += 1
|
|
255
|
+
if step.get("kind") == "llm" and not step.get("denied"):
|
|
256
|
+
st["llm"] += 1
|
|
257
|
+
st["cost"] += step.get("cost") or 0.0
|
|
258
|
+
lim = self.limits
|
|
259
|
+
reason = None
|
|
260
|
+
if lim["max_repeated_denials"] and st["streak"] >= lim["max_repeated_denials"]:
|
|
261
|
+
reason = f"repeated_denials: {st['last']} refused {st['streak']}x in a row (last rule {step.get('rule')})"
|
|
262
|
+
elif lim["max_denials"] and st["denials"] >= lim["max_denials"]:
|
|
263
|
+
reason = f"denials: {st['denials']} denials in one run"
|
|
264
|
+
elif lim["max_node_visits"] and st["nodes"] and max(st["nodes"].values()) >= lim["max_node_visits"]:
|
|
265
|
+
node, n = st["nodes"].most_common(1)[0]
|
|
266
|
+
reason = f"loop: node '{node}' ran {n}x"
|
|
267
|
+
elif lim["max_run_cost_usd"] and st["cost"] >= lim["max_run_cost_usd"]:
|
|
268
|
+
reason = f"cost: run spent ${st['cost']:.4f} (limit ${lim['max_run_cost_usd']})"
|
|
269
|
+
elif lim["max_llm_calls"] and st["llm"] >= lim["max_llm_calls"]:
|
|
270
|
+
reason = f"llm_calls: {st['llm']} model calls"
|
|
271
|
+
if not reason:
|
|
272
|
+
return
|
|
273
|
+
st["tripped"] = True
|
|
274
|
+
grant = current_grant()
|
|
275
|
+
self.trips.append({"run_id": run.id, "reason": reason, "agent": getattr(grant, "agent_name", None), "ts": time.time()})
|
|
276
|
+
if self.action == "revoke" and grant is not None and self.kernel is not None:
|
|
277
|
+
self.kernel.revoke(grant, reason=f"agentdynamics.watchdog: {reason}")
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
# ---------------------------------------------------------------- entry point
|
|
281
|
+
|
|
282
|
+
class Governance:
|
|
283
|
+
def __init__(self, kernel, root, gate, watchdog, unregister):
|
|
284
|
+
self.kernel, self.root, self.gate, self.watchdog, self._unregister = kernel, root, gate, watchdog, unregister
|
|
285
|
+
|
|
286
|
+
def uninstall(self):
|
|
287
|
+
self._unregister()
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def instrument(kernel, root=None, *, gate_models=True, watchdog=None, correlate=True, record_decisions=True):
|
|
291
|
+
"""Connect an Aegis kernel to AgentDynamics. Call after `agentdynamics.init()` and `build_kernel()`.
|
|
292
|
+
|
|
293
|
+
kernel the Aegis Kernel
|
|
294
|
+
root the root Grant (model calls outside a bound grant are charged to it)
|
|
295
|
+
gate_models reserve model spend against the Aegis budget before each call
|
|
296
|
+
watchdog a Watchdog, or None
|
|
297
|
+
"""
|
|
298
|
+
_require_aegis()
|
|
299
|
+
_state.update(kernel=kernel, root=root)
|
|
300
|
+
undo = []
|
|
301
|
+
if record_decisions:
|
|
302
|
+
_wrap_kernel(kernel)
|
|
303
|
+
if correlate:
|
|
304
|
+
try:
|
|
305
|
+
from aegis.observe import register_context_provider
|
|
306
|
+
except ImportError:
|
|
307
|
+
at._warn("aegis-obs", "this aegis version has no aegis.observe; decisions are recorded but not correlated")
|
|
308
|
+
else:
|
|
309
|
+
undo.append(register_context_provider(_context))
|
|
310
|
+
gate = None
|
|
311
|
+
if gate_models:
|
|
312
|
+
if hasattr(kernel, "reserve_spend"):
|
|
313
|
+
gate = ModelSpendGate(kernel)
|
|
314
|
+
at._hooks["llm_gates"].append(gate)
|
|
315
|
+
undo.append(lambda: at._hooks["llm_gates"].remove(gate))
|
|
316
|
+
else:
|
|
317
|
+
at._warn("aegis-old", "this aegis version has no reserve_spend; model spend is observed but not gated")
|
|
318
|
+
def tag_agent(run, step): # model calls don't pass through the kernel; attribute them to the bound agent
|
|
319
|
+
if step.get("kind") == "llm" and not step.get("agent"):
|
|
320
|
+
g = current_grant()
|
|
321
|
+
if g is not None:
|
|
322
|
+
step["agent"] = g.agent_name
|
|
323
|
+
step["governed"] = True
|
|
324
|
+
at._hooks["step"].insert(0, tag_agent)
|
|
325
|
+
undo.append(lambda: at._hooks["step"].remove(tag_agent))
|
|
326
|
+
if watchdog is not None:
|
|
327
|
+
watchdog.kernel = kernel
|
|
328
|
+
at._hooks["step"].append(watchdog)
|
|
329
|
+
undo.append(lambda: at._hooks["step"].remove(watchdog))
|
|
330
|
+
|
|
331
|
+
info_cache = {}
|
|
332
|
+
|
|
333
|
+
def meta(run):
|
|
334
|
+
grant = current_grant()
|
|
335
|
+
policy = getattr(grant, "policy", None)
|
|
336
|
+
if policy is None:
|
|
337
|
+
return {}
|
|
338
|
+
key = id(policy)
|
|
339
|
+
if key not in info_cache:
|
|
340
|
+
info_cache[key] = policy_info(policy)
|
|
341
|
+
info = info_cache[key]
|
|
342
|
+
return {"policy_version": info["label"], "policy": info, "framework": "agentdynamics-sdk+aegis"}
|
|
343
|
+
at._hooks["run_meta"].append(meta)
|
|
344
|
+
undo.append(lambda: at._hooks["run_meta"].remove(meta))
|
|
345
|
+
|
|
346
|
+
def unregister():
|
|
347
|
+
for u in undo:
|
|
348
|
+
try:
|
|
349
|
+
u()
|
|
350
|
+
except ValueError:
|
|
351
|
+
pass
|
|
352
|
+
return Governance(kernel, root, gate, watchdog, unregister)
|
agentdynamics/phases.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""Classify agent actions into process phases.
|
|
2
|
+
|
|
3
|
+
Phases model *how* an agent works a task, the way AppDynamics splits a
|
|
4
|
+
transaction into tiers: explore -> plan -> edit -> verify, plus delegate,
|
|
5
|
+
communicate and other.
|
|
6
|
+
"""
|
|
7
|
+
import re
|
|
8
|
+
|
|
9
|
+
EXPLORE_TOOLS = {
|
|
10
|
+
"Read", "Grep", "Glob", "LS", "WebSearch", "WebFetch", "ToolSearch", "NotebookRead",
|
|
11
|
+
"ListMcpResourcesTool", "ReadMcpResourceTool",
|
|
12
|
+
}
|
|
13
|
+
PLAN_TOOLS = {
|
|
14
|
+
"TodoWrite", "TaskCreate", "TaskUpdate", "TaskList", "EnterPlanMode", "ExitPlanMode",
|
|
15
|
+
"AskUserQuestion", "Skill",
|
|
16
|
+
}
|
|
17
|
+
EDIT_TOOLS = {"Edit", "Write", "MultiEdit", "NotebookEdit"}
|
|
18
|
+
DELEGATE_TOOLS = {"Agent", "Task", "Workflow", "SendMessage", "TaskOutput", "TaskStop"}
|
|
19
|
+
COMMUNICATE_TOOLS = {"Artifact", "SendUserFile", "PushNotification"}
|
|
20
|
+
SHELL_TOOLS = {"Bash", "PowerShell"}
|
|
21
|
+
|
|
22
|
+
VERIFY_CMD = re.compile(
|
|
23
|
+
r"\b(pytest|unittest|jest|vitest|mocha|go test|cargo (test|build|check|clippy)|npm (run )?(test|build|lint)|"
|
|
24
|
+
r"yarn (test|build)|pnpm (test|build)|tsc\b|mypy|ruff|eslint|flake8|pylint|make( test)?\b|mvn|gradle|dotnet (test|build)|"
|
|
25
|
+
r"python[0-9.]* -m (pytest|unittest|py_compile)|node --check|py_compile|curl\b|Invoke-WebRequest|playwright)"
|
|
26
|
+
)
|
|
27
|
+
RUN_CMD = re.compile(r"\b(python[0-9.]*|node|npm|npx|deno|go run|cargo run|uvicorn|flask|docker)\b")
|
|
28
|
+
EXPLORE_CMD = re.compile(
|
|
29
|
+
r"^\s*(cd [^&;]+(&&|;)\s*)?(ls|dir|cat|head|tail|find|grep|rg|tree|wc|pwd|echo|type|Get-ChildItem|Get-Content|"
|
|
30
|
+
r"Select-String|git (status|log|diff|show|branch|remote)|du|stat|which|where)\b"
|
|
31
|
+
)
|
|
32
|
+
EDIT_CMD = re.compile(r"(\bsed -i\b|>\s*[\w./\\-]+\.\w+\s*<<|\bcat\s*>|Set-Content|Out-File|\bmv\b|\bcp\b|\brm\b|mkdir)")
|
|
33
|
+
ADHOC_CMD = re.compile(r"\b(python[0-9.]*\s+(-\s|-c\b|-\s*<<)|node -e\b)")
|
|
34
|
+
GIT_CMD =re.compile(r"\bgit (add|commit|push|pull|merge|rebase|checkout|init|clone)|\bgh\b")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def classify(tool_name, tool_input):
|
|
38
|
+
"""Return (phase, target) for a tool call."""
|
|
39
|
+
inp = tool_input if isinstance(tool_input, dict) else {}
|
|
40
|
+
name = tool_name or ""
|
|
41
|
+
target = (
|
|
42
|
+
inp.get("file_path") or inp.get("notebook_path") or inp.get("path") or inp.get("url")
|
|
43
|
+
or inp.get("pattern") or inp.get("query") or inp.get("command") or inp.get("description")
|
|
44
|
+
or inp.get("skill") or ""
|
|
45
|
+
)
|
|
46
|
+
target = str(target)[:300]
|
|
47
|
+
if name in EDIT_TOOLS:
|
|
48
|
+
return "edit", target
|
|
49
|
+
if name in EXPLORE_TOOLS:
|
|
50
|
+
return "explore", target
|
|
51
|
+
if name in PLAN_TOOLS:
|
|
52
|
+
return "plan", target
|
|
53
|
+
if name in DELEGATE_TOOLS:
|
|
54
|
+
return "delegate", target
|
|
55
|
+
if name in COMMUNICATE_TOOLS or name.startswith("mcp__visualize"):
|
|
56
|
+
return "communicate", target
|
|
57
|
+
if name in SHELL_TOOLS:
|
|
58
|
+
cmd = str(inp.get("command", ""))
|
|
59
|
+
if ADHOC_CMD.search(cmd):
|
|
60
|
+
return "execute", target
|
|
61
|
+
if VERIFY_CMD.search(cmd):
|
|
62
|
+
return "verify", target
|
|
63
|
+
if GIT_CMD.search(cmd):
|
|
64
|
+
return "vcs", target
|
|
65
|
+
if EXPLORE_CMD.search(cmd):
|
|
66
|
+
return "explore", target
|
|
67
|
+
if EDIT_CMD.search(cmd):
|
|
68
|
+
return "edit", target
|
|
69
|
+
if RUN_CMD.search(cmd):
|
|
70
|
+
return "verify", target
|
|
71
|
+
return "execute", target
|
|
72
|
+
if name.startswith("mcp__"):
|
|
73
|
+
low = name.lower()
|
|
74
|
+
if any(k in low for k in ("screenshot", "preview", "console", "read_page", "get_page_text", "network", "find", "logs")):
|
|
75
|
+
return "verify", target
|
|
76
|
+
if any(k in low for k in ("navigate", "computer", "click", "form_input", "javascript")):
|
|
77
|
+
return "verify", target
|
|
78
|
+
return "execute", target
|
|
79
|
+
return "other", target
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
PHASES = ["explore", "plan", "edit", "verify", "execute", "vcs", "delegate", "communicate", "other"]
|