agentdynamics 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agentdynamics/__init__.py +12 -0
- agentdynamics/__main__.py +322 -0
- agentdynamics/analysis.py +755 -0
- agentdynamics/autotrace.py +697 -0
- agentdynamics/bootstrap/sitecustomize.py +27 -0
- agentdynamics/collectors/__init__.py +0 -0
- agentdynamics/collectors/aegis_audit.py +72 -0
- agentdynamics/collectors/claude_code.py +257 -0
- agentdynamics/collectors/generic.py +103 -0
- agentdynamics/collectors/inbox.py +95 -0
- agentdynamics/collectors/langfuse.py +98 -0
- agentdynamics/collectors/langsmith.py +258 -0
- agentdynamics/collectors/otlp.py +349 -0
- agentdynamics/collectors/spans.py +202 -0
- agentdynamics/config.py +139 -0
- agentdynamics/engine.py +376 -0
- agentdynamics/flows.py +142 -0
- agentdynamics/govern.py +231 -0
- agentdynamics/integrations/__init__.py +1 -0
- agentdynamics/integrations/aegis.py +352 -0
- agentdynamics/phases.py +82 -0
- agentdynamics/pricing.py +69 -0
- agentdynamics/privacy.py +41 -0
- agentdynamics/sdk.py +105 -0
- agentdynamics/server.py +949 -0
- agentdynamics/slo.py +84 -0
- agentdynamics/store.py +228 -0
- agentdynamics/web/app.js +1069 -0
- agentdynamics/web/charts.js +185 -0
- agentdynamics/web/index.html +40 -0
- agentdynamics/web/style.css +244 -0
- agentdynamics-0.4.0.dist-info/METADATA +195 -0
- agentdynamics-0.4.0.dist-info/RECORD +37 -0
- agentdynamics-0.4.0.dist-info/WHEEL +5 -0
- agentdynamics-0.4.0.dist-info/entry_points.txt +2 -0
- agentdynamics-0.4.0.dist-info/licenses/LICENSE +202 -0
- agentdynamics-0.4.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Loaded automatically by `agentdynamics run <command>` (placed first on PYTHONPATH).
|
|
2
|
+
|
|
3
|
+
Instruments the child Python process before any user code runs, like `ddtrace-run` or `opentelemetry-instrument`.
|
|
4
|
+
"""
|
|
5
|
+
import os
|
|
6
|
+
import sys
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
import agentdynamics
|
|
10
|
+
|
|
11
|
+
agentdynamics.init(quiet=os.environ.get("AGENTDYNAMICS_QUIET") == "1")
|
|
12
|
+
except Exception as ex: # never break the user's program
|
|
13
|
+
print(f"[agentdynamics] auto-instrumentation disabled: {ex}", file=sys.stderr)
|
|
14
|
+
|
|
15
|
+
# Chain to any sitecustomize that was shadowed by ours.
|
|
16
|
+
_here = os.path.dirname(os.path.abspath(__file__))
|
|
17
|
+
for _p in sys.path:
|
|
18
|
+
if _p and os.path.abspath(_p) != _here and os.path.isfile(os.path.join(_p, "sitecustomize.py")):
|
|
19
|
+
import importlib.util
|
|
20
|
+
|
|
21
|
+
_spec = importlib.util.spec_from_file_location("_user_sitecustomize", os.path.join(_p, "sitecustomize.py"))
|
|
22
|
+
_mod = importlib.util.module_from_spec(_spec)
|
|
23
|
+
try:
|
|
24
|
+
_spec.loader.exec_module(_mod)
|
|
25
|
+
except Exception as ex:
|
|
26
|
+
print(f"[agentdynamics] user sitecustomize failed: {ex}", file=sys.stderr)
|
|
27
|
+
break
|
|
File without changes
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Aegis audit logs -> AgentDynamics runs (out-of-process path).
|
|
2
|
+
|
|
3
|
+
For agents that run Aegis without the in-process integration (a different language runtime, CI
|
|
4
|
+
conformance runs, historical logs), point the inbox at the JSONL audit file (`AuditLog(path=...)`)
|
|
5
|
+
or POST records to /api/ingest/records. Records are grouped into runs by the correlation id Aegis
|
|
6
|
+
stamps into `details.ctx` (run_id / trace_id) when a context provider is registered, otherwise by
|
|
7
|
+
grant id.
|
|
8
|
+
|
|
9
|
+
Aegis records hash the arguments rather than storing them, so these runs show *what was decided*
|
|
10
|
+
(tool, verdict, rule, agent, depth) but not argument values. Policy export needs argument values and
|
|
11
|
+
therefore uses the in-process integration.
|
|
12
|
+
"""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
REQUIRED = {"seq", "grant_id", "prev_hash", "rule", "allowed", "tool"}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def is_record(rec):
|
|
19
|
+
return isinstance(rec, dict) and REQUIRED <= set(rec)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def group_key(rec):
|
|
23
|
+
ctx = (rec.get("details") or {}).get("ctx") or {}
|
|
24
|
+
return ctx.get("run_id") or ctx.get("trace_id") or f"grant:{rec.get('grant_id')}"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def span_id(rec):
|
|
28
|
+
return rec.get("hash") or f"{rec.get('grant_id')}:{rec.get('seq')}:{rec.get('ts')}"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def build_payload(group, records):
|
|
32
|
+
"""Generic-run payload for one group of audit records."""
|
|
33
|
+
records = sorted(records, key=lambda r: (r.get("ts") or 0, r.get("seq") or 0))
|
|
34
|
+
ctx = next(((r.get("details") or {}).get("ctx") for r in records if (r.get("details") or {}).get("ctx")), {}) or {}
|
|
35
|
+
t0 = records[0].get("ts")
|
|
36
|
+
steps = [{"kind": "prompt", "ts": t0, "text": ctx.get("workflow") or f"Aegis-governed run ({records[0].get('agent')})"}]
|
|
37
|
+
pending = {} # (grant, tool, args_digest) -> index of an admitted step a later record may deny
|
|
38
|
+
for r in records:
|
|
39
|
+
tool, rule, ts = r.get("tool"), r.get("rule"), r.get("ts")
|
|
40
|
+
det = r.get("details") or {}
|
|
41
|
+
base = {"ts": ts, "end_ts": ts, "agent": r.get("agent"), "grant_depth": r.get("depth"), "governed": True,
|
|
42
|
+
"rule": rule, "guard": r.get("guard"), "node": (det.get("ctx") or {}).get("node")}
|
|
43
|
+
if tool == "model.spend":
|
|
44
|
+
if rule == "budget.settled":
|
|
45
|
+
steps.append({**base, "kind": "llm", "model": "model", "cost": float(det.get("usd") or 0),
|
|
46
|
+
"input_tokens": int(det.get("tokens") or 0), "output_tokens": 0})
|
|
47
|
+
elif not r.get("allowed"):
|
|
48
|
+
steps.append({**base, "kind": "llm", "model": "model", "denied": True, "error": r.get("reason")})
|
|
49
|
+
continue
|
|
50
|
+
if tool == "agent.revoke":
|
|
51
|
+
steps.append({"kind": "notice", "name": "revoked", "ts": ts, "agent": r.get("agent"), "rule": rule,
|
|
52
|
+
"text": f"{r.get('agent')}: revoked"})
|
|
53
|
+
continue
|
|
54
|
+
key = (r.get("grant_id"), tool, r.get("args_digest"))
|
|
55
|
+
if not r.get("allowed") and key in pending:
|
|
56
|
+
# a budget charge or post-guard refused a call that had been admitted: one call, denied
|
|
57
|
+
i = pending.pop(key)
|
|
58
|
+
steps[i].update(denied=True, rule=rule, guard=r.get("guard"), error=f"[{rule}] {r.get('reason') or ''}")
|
|
59
|
+
continue
|
|
60
|
+
if tool == "agent.spawn" and r.get("allowed"):
|
|
61
|
+
steps.append({**base, "kind": "span", "span_kind": "agent", "name": "sub-agent", "rule": "spawn.granted",
|
|
62
|
+
"start_ts": ts})
|
|
63
|
+
continue
|
|
64
|
+
step = {**base, "kind": "tool", "name": tool}
|
|
65
|
+
if r.get("allowed"):
|
|
66
|
+
pending[key] = len(steps)
|
|
67
|
+
else:
|
|
68
|
+
step.update(denied=True, error=f"[{rule}] {r.get('reason') or ''}")
|
|
69
|
+
steps.append(step)
|
|
70
|
+
return {"id": f"aegis-{group}", "source": "aegis", "workflow": ctx.get("workflow") or "aegis-governed",
|
|
71
|
+
"project": ctx.get("project") or "aegis", "agent": records[0].get("agent"), "framework": "aegis",
|
|
72
|
+
"status": "ok", "complete": True, "steps": steps}
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Collector for Claude Code session transcripts (~/.claude/projects/**/*.jsonl).
|
|
2
|
+
|
|
3
|
+
Normalizes each transcript file into a *run*: session metadata plus an ordered
|
|
4
|
+
list of steps (prompt / llm / tool / notice). Analysis never looks at the raw
|
|
5
|
+
transcript format, so other agent frameworks only need their own collector.
|
|
6
|
+
"""
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
from datetime import datetime
|
|
11
|
+
|
|
12
|
+
from .. import pricing
|
|
13
|
+
from ..phases import classify
|
|
14
|
+
|
|
15
|
+
DEFAULT_ROOT = os.path.join(os.path.expanduser("~"), ".claude", "projects")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def parse_ts(s):
|
|
19
|
+
if not s:
|
|
20
|
+
return None
|
|
21
|
+
try:
|
|
22
|
+
return datetime.fromisoformat(s.replace("Z", "+00:00")).timestamp()
|
|
23
|
+
except ValueError:
|
|
24
|
+
return None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _content_len(c):
|
|
28
|
+
if c is None:
|
|
29
|
+
return 0
|
|
30
|
+
if isinstance(c, str):
|
|
31
|
+
return len(c)
|
|
32
|
+
if isinstance(c, list):
|
|
33
|
+
n = 0
|
|
34
|
+
for b in c:
|
|
35
|
+
if isinstance(b, dict):
|
|
36
|
+
if b.get("type") == "text":
|
|
37
|
+
n += len(b.get("text", ""))
|
|
38
|
+
elif b.get("type") == "image":
|
|
39
|
+
n += 6000 # ~1.5k tokens per image, rough
|
|
40
|
+
else:
|
|
41
|
+
n += len(json.dumps(b))
|
|
42
|
+
return n
|
|
43
|
+
return len(json.dumps(c))
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _text_of(c):
|
|
47
|
+
if isinstance(c, str):
|
|
48
|
+
return c
|
|
49
|
+
if isinstance(c, list):
|
|
50
|
+
return "\n".join(b.get("text", "") for b in c if isinstance(b, dict) and b.get("type") == "text")
|
|
51
|
+
return ""
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _classify_user_text(text, entry):
|
|
55
|
+
"""Return (kind, subkind) for a user text message."""
|
|
56
|
+
t = text.lstrip()
|
|
57
|
+
origin = (entry.get("origin") or {}).get("kind")
|
|
58
|
+
if entry.get("isCompactSummary") or t.startswith("This session is being continued from a previous conversation"):
|
|
59
|
+
return "notice", "compaction"
|
|
60
|
+
if t.startswith("[Request interrupted by user"):
|
|
61
|
+
return "notice", "interrupt"
|
|
62
|
+
if origin == "task-notification" or t.startswith("<task-notification"):
|
|
63
|
+
return "notice", "task_notification"
|
|
64
|
+
if t.startswith("<local-command-stdout") or t.startswith("<local-command-caveat") or t.startswith("<local-command-stderr"):
|
|
65
|
+
return "skip", None
|
|
66
|
+
if entry.get("isMeta"):
|
|
67
|
+
return "skip", None
|
|
68
|
+
if t.startswith("Base directory for this skill") or t.startswith("<system-reminder>"):
|
|
69
|
+
return "skip", None
|
|
70
|
+
if t.startswith("<scheduled-task"):
|
|
71
|
+
return "prompt", "scheduled"
|
|
72
|
+
if t.startswith("<command-name>") or t.startswith("<command-message>"):
|
|
73
|
+
return "prompt", "command"
|
|
74
|
+
return "prompt", "human"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _hash_input(name, inp):
|
|
78
|
+
return hashlib.sha1((name + json.dumps(inp, sort_keys=True, default=str)).encode()).hexdigest()[:16]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def parse_file(path, root=DEFAULT_ROOT):
|
|
82
|
+
rel = os.path.relpath(path, root)
|
|
83
|
+
parts = rel.replace("\\", "/").split("/")
|
|
84
|
+
project_dir = parts[0]
|
|
85
|
+
is_sub = "subagents" in parts
|
|
86
|
+
stem = os.path.splitext(parts[-1])[0]
|
|
87
|
+
parent_id = parts[1] if is_sub else None
|
|
88
|
+
run = {
|
|
89
|
+
"id": stem if not is_sub else f"{parent_id}:{stem}",
|
|
90
|
+
"source": "claude-code",
|
|
91
|
+
"file": path,
|
|
92
|
+
"project": project_dir,
|
|
93
|
+
"cwd": None,
|
|
94
|
+
"title": None,
|
|
95
|
+
"agent_name": None,
|
|
96
|
+
"parent_id": parent_id,
|
|
97
|
+
"is_subagent": is_sub,
|
|
98
|
+
"workflow": parts[3] if is_sub and len(parts) > 4 and parts[2] == "workflows" else None,
|
|
99
|
+
"version": None,
|
|
100
|
+
"git_branch": None,
|
|
101
|
+
"entrypoint": None,
|
|
102
|
+
"steps": [],
|
|
103
|
+
}
|
|
104
|
+
steps = run["steps"]
|
|
105
|
+
llm_by_msg = {}
|
|
106
|
+
tool_by_id = {}
|
|
107
|
+
last_ts = None
|
|
108
|
+
|
|
109
|
+
with open(path, encoding="utf-8", errors="replace") as f:
|
|
110
|
+
for line in f:
|
|
111
|
+
try:
|
|
112
|
+
e = json.loads(line)
|
|
113
|
+
except json.JSONDecodeError:
|
|
114
|
+
continue
|
|
115
|
+
et = e.get("type")
|
|
116
|
+
if et == "custom-title":
|
|
117
|
+
run["title"] = e.get("customTitle")
|
|
118
|
+
continue
|
|
119
|
+
if et == "agent-name":
|
|
120
|
+
run["agent_name"] = e.get("agentName")
|
|
121
|
+
continue
|
|
122
|
+
ts = parse_ts(e.get("timestamp"))
|
|
123
|
+
if e.get("cwd") and not run["cwd"]:
|
|
124
|
+
run["cwd"] = e["cwd"]
|
|
125
|
+
run["version"] = e.get("version")
|
|
126
|
+
run["git_branch"] = e.get("gitBranch")
|
|
127
|
+
run["entrypoint"] = e.get("entrypoint")
|
|
128
|
+
|
|
129
|
+
if et == "system":
|
|
130
|
+
st = e.get("subtype")
|
|
131
|
+
if st in ("api_error", "compact_boundary"):
|
|
132
|
+
steps.append({"kind": "notice", "name": st if st != "compact_boundary" else "compaction",
|
|
133
|
+
"ts": ts, "text": str(e.get("content") or e.get("error") or "")[:300]})
|
|
134
|
+
if ts:
|
|
135
|
+
last_ts = ts
|
|
136
|
+
continue
|
|
137
|
+
|
|
138
|
+
msg = e.get("message")
|
|
139
|
+
if et == "assistant" and isinstance(msg, dict):
|
|
140
|
+
mid = msg.get("id") or e.get("uuid")
|
|
141
|
+
step = llm_by_msg.get(mid)
|
|
142
|
+
usage = msg.get("usage") or {}
|
|
143
|
+
if step is None:
|
|
144
|
+
step = {
|
|
145
|
+
"kind": "llm", "name": msg.get("model"), "model": msg.get("model"), "ts": ts,
|
|
146
|
+
"start_ts": last_ts if last_ts and ts and ts - last_ts < 3600 else ts,
|
|
147
|
+
"end_ts": ts, "text_chars": 0, "thinking_blocks": 0, "tool_calls": 0,
|
|
148
|
+
"sidechain": bool(e.get("isSidechain")), "text": "",
|
|
149
|
+
}
|
|
150
|
+
llm_by_msg[mid] = step
|
|
151
|
+
steps.append(step)
|
|
152
|
+
step["end_ts"] = ts or step["end_ts"]
|
|
153
|
+
step["stop_reason"] = msg.get("stop_reason") or step.get("stop_reason")
|
|
154
|
+
cc = usage.get("cache_creation") or {}
|
|
155
|
+
cw_total = usage.get("cache_creation_input_tokens", 0) or 0
|
|
156
|
+
cw_1h = cc.get("ephemeral_1h_input_tokens", 0) or 0
|
|
157
|
+
cw_5m = cc.get("ephemeral_5m_input_tokens", cw_total - cw_1h) or 0
|
|
158
|
+
step.update({
|
|
159
|
+
"input_tokens": usage.get("input_tokens", 0) or 0,
|
|
160
|
+
"output_tokens": usage.get("output_tokens", 0) or 0,
|
|
161
|
+
"cache_read": usage.get("cache_read_input_tokens", 0) or 0,
|
|
162
|
+
"cache_write": cw_total,
|
|
163
|
+
"cache_write_1h": cw_1h,
|
|
164
|
+
"cache_write_5m": cw_5m,
|
|
165
|
+
"thinking_tokens": (usage.get("output_tokens_details") or {}).get("thinking_tokens", 0) or 0,
|
|
166
|
+
"effort": e.get("effort") or step.get("effort"),
|
|
167
|
+
})
|
|
168
|
+
for b in msg.get("content") or []:
|
|
169
|
+
bt = b.get("type")
|
|
170
|
+
if bt == "text":
|
|
171
|
+
step["text_chars"] += len(b.get("text", ""))
|
|
172
|
+
if len(step["text"]) < 600:
|
|
173
|
+
step["text"] += b.get("text", "")[:600]
|
|
174
|
+
elif bt == "thinking":
|
|
175
|
+
step["thinking_blocks"] += 1
|
|
176
|
+
elif bt == "tool_use":
|
|
177
|
+
step["tool_calls"] += 1
|
|
178
|
+
name = b.get("name")
|
|
179
|
+
inp = b.get("input") or {}
|
|
180
|
+
phase, target = classify(name, inp)
|
|
181
|
+
tstep = {
|
|
182
|
+
"kind": "tool", "name": name, "ts": ts, "start_ts": ts, "end_ts": None,
|
|
183
|
+
"phase": phase, "target": target, "input_hash": _hash_input(name, inp),
|
|
184
|
+
"input_chars": len(json.dumps(inp)), "llm_msg": mid, "is_error": False,
|
|
185
|
+
"output_chars": 0, "tool_use_id": b.get("id"),
|
|
186
|
+
"input_preview": json.dumps(inp)[:400],
|
|
187
|
+
}
|
|
188
|
+
tool_by_id[b.get("id")] = tstep
|
|
189
|
+
steps.append(tstep)
|
|
190
|
+
if ts:
|
|
191
|
+
last_ts = ts
|
|
192
|
+
continue
|
|
193
|
+
|
|
194
|
+
if et == "user" and isinstance(msg, dict):
|
|
195
|
+
content = msg.get("content")
|
|
196
|
+
blocks = content if isinstance(content, list) else []
|
|
197
|
+
tool_results = [b for b in blocks if isinstance(b, dict) and b.get("type") == "tool_result"]
|
|
198
|
+
if tool_results:
|
|
199
|
+
for b in tool_results:
|
|
200
|
+
tstep = tool_by_id.get(b.get("tool_use_id"))
|
|
201
|
+
if not tstep:
|
|
202
|
+
continue
|
|
203
|
+
tstep["end_ts"] = ts
|
|
204
|
+
tstep["is_error"] = bool(b.get("is_error"))
|
|
205
|
+
tstep["output_chars"] = _content_len(b.get("content"))
|
|
206
|
+
if tstep["is_error"]:
|
|
207
|
+
tstep["error"] = _text_of(b.get("content"))[:300] or str(b.get("content"))[:300]
|
|
208
|
+
tur = e.get("toolUseResult")
|
|
209
|
+
if isinstance(tur, dict):
|
|
210
|
+
if tur.get("interrupted"):
|
|
211
|
+
tstep["interrupted"] = True
|
|
212
|
+
if tur.get("agentId"):
|
|
213
|
+
tstep["subagent_id"] = tur.get("agentId")
|
|
214
|
+
tstep["subagent_tokens"] = tur.get("totalTokens")
|
|
215
|
+
elif isinstance(tur, str) and "rejected" in tur.lower():
|
|
216
|
+
tstep["rejected"] = True
|
|
217
|
+
if ts:
|
|
218
|
+
last_ts = ts
|
|
219
|
+
continue
|
|
220
|
+
text = _text_of(content)
|
|
221
|
+
if not text.strip():
|
|
222
|
+
continue
|
|
223
|
+
kind, sub = _classify_user_text(text, e)
|
|
224
|
+
if kind == "skip":
|
|
225
|
+
continue
|
|
226
|
+
steps.append({"kind": kind, "name": sub, "ts": ts, "text": text[:2000], "sidechain": bool(e.get("isSidechain"))})
|
|
227
|
+
if ts:
|
|
228
|
+
last_ts = ts
|
|
229
|
+
|
|
230
|
+
# finalize costs and durations
|
|
231
|
+
for s in steps:
|
|
232
|
+
if s["kind"] == "llm":
|
|
233
|
+
s["cost"] = pricing.cost(s["model"], s.get("input_tokens", 0), s.get("output_tokens", 0),
|
|
234
|
+
s.get("cache_read", 0), s.get("cache_write_5m", 0), s.get("cache_write_1h", 0))
|
|
235
|
+
s["context_tokens"] = s.get("input_tokens", 0) + s.get("cache_read", 0) + s.get("cache_write", 0)
|
|
236
|
+
if s.get("start_ts") and s.get("end_ts"):
|
|
237
|
+
s["duration_ms"] = max(0, int((s["end_ts"] - s["start_ts"]) * 1000))
|
|
238
|
+
elif s["kind"] == "tool":
|
|
239
|
+
if s.get("end_ts") and s.get("start_ts"):
|
|
240
|
+
s["duration_ms"] = max(0, int((s["end_ts"] - s["start_ts"]) * 1000))
|
|
241
|
+
for i, s in enumerate(steps):
|
|
242
|
+
s["seq"] = i
|
|
243
|
+
tss = [s["ts"] for s in steps if s.get("ts")]
|
|
244
|
+
run["started"] = min(tss) if tss else None
|
|
245
|
+
run["ended"] = max(tss) if tss else None
|
|
246
|
+
if is_sub and steps and steps[0]["kind"] == "prompt":
|
|
247
|
+
steps[0]["name"] = "delegated"
|
|
248
|
+
return run
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def discover(root=DEFAULT_ROOT):
|
|
252
|
+
out = []
|
|
253
|
+
for dirpath, _dirs, files in os.walk(root):
|
|
254
|
+
for fn in files:
|
|
255
|
+
if fn.endswith(".jsonl"):
|
|
256
|
+
out.append(os.path.join(dirpath, fn))
|
|
257
|
+
return out
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Collector for runs pushed by any agent framework (SDK / HTTP ingest).
|
|
2
|
+
|
|
3
|
+
Accepted format (JSON), one run per file in <data_dir>/runs/:
|
|
4
|
+
{
|
|
5
|
+
"id": "run-123", "agent": "support-bot", "project": "helpdesk",
|
|
6
|
+
"steps": [
|
|
7
|
+
{"kind": "prompt", "ts": 1718000000.0, "text": "Refund order 42"},
|
|
8
|
+
{"kind": "llm", "ts": ..., "end_ts": ..., "model": "claude-opus-5",
|
|
9
|
+
"input_tokens": 1200, "output_tokens": 300, "cache_read": 0, "cache_write": 0, "stop_reason": "tool_use"},
|
|
10
|
+
{"kind": "tool", "ts": ..., "end_ts": ..., "name": "lookup_order", "input": {...},
|
|
11
|
+
"is_error": false, "output_chars": 512, "phase": "explore"}
|
|
12
|
+
]
|
|
13
|
+
}
|
|
14
|
+
Missing costs are computed from pricing; missing phases are inferred.
|
|
15
|
+
"""
|
|
16
|
+
import hashlib
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import uuid
|
|
20
|
+
|
|
21
|
+
from .. import pricing
|
|
22
|
+
from ..phases import classify
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def normalize(payload):
|
|
26
|
+
rid = str(payload.get("id") or uuid.uuid4())
|
|
27
|
+
run = {
|
|
28
|
+
"id": rid, "source": payload.get("source") or "sdk", "file": None,
|
|
29
|
+
"project": payload.get("project") or payload.get("agent") or "default",
|
|
30
|
+
"cwd": payload.get("cwd"), "title": payload.get("title"), "agent_name": payload.get("agent"),
|
|
31
|
+
"parent_id": payload.get("parent_id"), "is_subagent": bool(payload.get("parent_id")),
|
|
32
|
+
"workflow": payload.get("workflow"), "version": payload.get("version"), "git_branch": None,
|
|
33
|
+
"entrypoint": payload.get("entrypoint"), "environment": payload.get("environment") or "default",
|
|
34
|
+
"framework": payload.get("framework") or "sdk", "thread_id": payload.get("thread_id"), "user_id": payload.get("user_id"),
|
|
35
|
+
"feedback": payload.get("feedback") or [], "tags": payload.get("tags") or [],
|
|
36
|
+
"root_status": payload.get("status"), "root_error": payload.get("error"), "complete": payload.get("complete", True),
|
|
37
|
+
"policy_version": payload.get("policy_version"), "policy": payload.get("policy"),
|
|
38
|
+
"steps": [],
|
|
39
|
+
}
|
|
40
|
+
last_llm = None
|
|
41
|
+
for i, s in enumerate(payload.get("steps") or []):
|
|
42
|
+
k = s.get("kind")
|
|
43
|
+
st = dict(s)
|
|
44
|
+
st["seq"] = i
|
|
45
|
+
st.setdefault("ts", st.get("start_ts"))
|
|
46
|
+
st.setdefault("start_ts", st.get("ts"))
|
|
47
|
+
if st.get("ts") and st.get("end_ts") and "duration_ms" not in st:
|
|
48
|
+
st["duration_ms"] = int((st["end_ts"] - st["ts"]) * 1000)
|
|
49
|
+
if k == "llm":
|
|
50
|
+
st.setdefault("name", st.get("model"))
|
|
51
|
+
for f in ("input_tokens", "output_tokens", "cache_read", "cache_write", "thinking_tokens"):
|
|
52
|
+
st[f] = int(st.get(f) or 0)
|
|
53
|
+
st.setdefault("cache_write_5m", st["cache_write"])
|
|
54
|
+
st.setdefault("cache_write_1h", 0)
|
|
55
|
+
if st.get("cost") is None:
|
|
56
|
+
st["cost"] = pricing.cost(st.get("model"), st["input_tokens"], st["output_tokens"], st["cache_read"],
|
|
57
|
+
st["cache_write_5m"], st["cache_write_1h"])
|
|
58
|
+
st["context_tokens"] = st["input_tokens"] + st["cache_read"] + st["cache_write"]
|
|
59
|
+
st.setdefault("text", "")
|
|
60
|
+
st["_id"] = st.get("id") or f"{rid}:{i}"
|
|
61
|
+
st["tool_calls"] = 0
|
|
62
|
+
last_llm = st
|
|
63
|
+
elif k == "tool":
|
|
64
|
+
inp = st.pop("input", None) or {}
|
|
65
|
+
phase, target = classify(st.get("name"), inp)
|
|
66
|
+
st.setdefault("phase", phase)
|
|
67
|
+
st.setdefault("target", target)
|
|
68
|
+
st["input_hash"] = hashlib.sha1((str(st.get("name")) + json.dumps(inp, sort_keys=True, default=str)).encode()).hexdigest()[:16]
|
|
69
|
+
st["input_preview"] = json.dumps(inp, default=str)[:400]
|
|
70
|
+
if inp:
|
|
71
|
+
st["args_json"] = json.dumps(inp, default=str)[:4000]
|
|
72
|
+
st["is_error"] = bool(st.get("is_error"))
|
|
73
|
+
st["output_chars"] = int(st.get("output_chars") or 0)
|
|
74
|
+
if last_llm is not None:
|
|
75
|
+
st["llm_msg"] = last_llm["_id"]
|
|
76
|
+
last_llm["tool_calls"] += 1
|
|
77
|
+
elif k == "prompt":
|
|
78
|
+
st.setdefault("name", "human")
|
|
79
|
+
elif k == "span":
|
|
80
|
+
st.setdefault("span_kind", "node")
|
|
81
|
+
st.setdefault("node", st.get("name"))
|
|
82
|
+
run["steps"].append(st)
|
|
83
|
+
tss = [s["ts"] for s in run["steps"] if s.get("ts")]
|
|
84
|
+
run["started"] = min(tss) if tss else None
|
|
85
|
+
run["ended"] = max(tss) if tss else None
|
|
86
|
+
return run
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def load_dir(runs_dir):
|
|
90
|
+
out = []
|
|
91
|
+
if not os.path.isdir(runs_dir):
|
|
92
|
+
return out
|
|
93
|
+
for fn in sorted(os.listdir(runs_dir)):
|
|
94
|
+
if fn.endswith(".json"):
|
|
95
|
+
p = os.path.join(runs_dir, fn)
|
|
96
|
+
try:
|
|
97
|
+
with open(p, encoding="utf-8") as f:
|
|
98
|
+
run = normalize(json.load(f))
|
|
99
|
+
run["file"] = p
|
|
100
|
+
out.append(run)
|
|
101
|
+
except (ValueError, OSError):
|
|
102
|
+
continue
|
|
103
|
+
return out
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
"""File inbox / log tailing: the bridge for log pipelines (Fluent Bit, Vector, Logstash, OTel Collector
|
|
2
|
+
file exporter, S3/GCS sync jobs...). Watches a directory of *.jsonl / *.json files, tails growing files
|
|
3
|
+
by byte offset, and auto-detects each record's format:
|
|
4
|
+
|
|
5
|
+
- OTLP JSON export {"resourceSpans": [...]} (OTel Collector `file` exporter)
|
|
6
|
+
- LangSmith run {"id", "run_type", "trace_id", ...} (LangSmith bulk export / run dumps)
|
|
7
|
+
- Langfuse trace {"id", "observations": [...]}
|
|
8
|
+
- AgentDynamics generic run {"steps": [...]}
|
|
9
|
+
- Canonical span {"trace_id", "span_id", "kind", ...}
|
|
10
|
+
"""
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def detect(rec):
|
|
16
|
+
if not isinstance(rec, dict):
|
|
17
|
+
return None
|
|
18
|
+
if "resourceSpans" in rec:
|
|
19
|
+
return "otlp"
|
|
20
|
+
if "run_type" in rec and ("trace_id" in rec or "id" in rec):
|
|
21
|
+
return "langsmith"
|
|
22
|
+
if "observations" in rec and "id" in rec:
|
|
23
|
+
return "langfuse"
|
|
24
|
+
if "steps" in rec:
|
|
25
|
+
return "generic"
|
|
26
|
+
if "trace_id" in rec and "span_id" in rec:
|
|
27
|
+
return "span"
|
|
28
|
+
if {"seq", "grant_id", "prev_hash", "rule", "allowed", "tool"} <= set(rec):
|
|
29
|
+
return "aegis"
|
|
30
|
+
return None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def unwrap(rec):
|
|
34
|
+
"""Log shippers wrap payloads: {"log": "<json>"}, {"message": {...}}, {"body": ...}."""
|
|
35
|
+
for key in ("log", "message", "body", "record"):
|
|
36
|
+
if detect(rec) is None and isinstance(rec, dict) and key in rec:
|
|
37
|
+
inner = rec[key]
|
|
38
|
+
if isinstance(inner, str):
|
|
39
|
+
try:
|
|
40
|
+
inner = json.loads(inner)
|
|
41
|
+
except ValueError:
|
|
42
|
+
continue
|
|
43
|
+
rec = inner
|
|
44
|
+
return rec
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class Inbox:
|
|
48
|
+
def __init__(self, path, state):
|
|
49
|
+
self.path = path
|
|
50
|
+
self.state = state # {"offsets": {file: bytes}}
|
|
51
|
+
os.makedirs(path, exist_ok=True)
|
|
52
|
+
|
|
53
|
+
def poll(self, max_bytes=64 * 1024 * 1024):
|
|
54
|
+
"""Return list of (format, record) read since last poll."""
|
|
55
|
+
offs = self.state.setdefault("offsets", {})
|
|
56
|
+
out = []
|
|
57
|
+
for fn in sorted(os.listdir(self.path)):
|
|
58
|
+
p = os.path.join(self.path, fn)
|
|
59
|
+
if not os.path.isfile(p) or not fn.endswith((".jsonl", ".json", ".ndjson", ".log")):
|
|
60
|
+
continue
|
|
61
|
+
size = os.path.getsize(p)
|
|
62
|
+
start = offs.get(p, 0)
|
|
63
|
+
if size < start: # rotated / truncated
|
|
64
|
+
start = 0
|
|
65
|
+
if size == start:
|
|
66
|
+
continue
|
|
67
|
+
with open(p, "rb") as f:
|
|
68
|
+
f.seek(start)
|
|
69
|
+
chunk = f.read(max_bytes)
|
|
70
|
+
if fn.endswith(".json") and start == 0:
|
|
71
|
+
try:
|
|
72
|
+
doc = json.loads(chunk)
|
|
73
|
+
recs = doc if isinstance(doc, list) else [doc]
|
|
74
|
+
out.extend((detect(r), r) for r in recs if detect(r))
|
|
75
|
+
offs[p] = size
|
|
76
|
+
continue
|
|
77
|
+
except ValueError:
|
|
78
|
+
pass
|
|
79
|
+
last_nl = chunk.rfind(b"\n")
|
|
80
|
+
if last_nl < 0:
|
|
81
|
+
continue # wait for a complete line
|
|
82
|
+
for line in chunk[:last_nl].splitlines():
|
|
83
|
+
line = line.strip()
|
|
84
|
+
if not line:
|
|
85
|
+
continue
|
|
86
|
+
try:
|
|
87
|
+
rec = json.loads(line)
|
|
88
|
+
except ValueError:
|
|
89
|
+
continue
|
|
90
|
+
rec = unwrap(rec)
|
|
91
|
+
fmt = detect(rec)
|
|
92
|
+
if fmt:
|
|
93
|
+
out.append((fmt, rec))
|
|
94
|
+
offs[p] = start + last_nl + 1
|
|
95
|
+
return out
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""Langfuse pull connector (public API) -> canonical spans.
|
|
2
|
+
|
|
3
|
+
Traces: GET /api/public/traces?fromTimestamp=..&page=..&limit=..
|
|
4
|
+
Detail: GET /api/public/traces/{id} (includes observations: SPAN | GENERATION | EVENT | AGENT | TOOL | RETRIEVER ...)
|
|
5
|
+
Auth: HTTP Basic public_key:secret_key
|
|
6
|
+
"""
|
|
7
|
+
import base64
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import time
|
|
11
|
+
import urllib.request
|
|
12
|
+
from datetime import datetime, timezone
|
|
13
|
+
from urllib.parse import urlencode
|
|
14
|
+
|
|
15
|
+
from .langsmith import parse_time
|
|
16
|
+
|
|
17
|
+
OBS_KIND = {"GENERATION": "llm", "SPAN": "chain", "EVENT": "span", "AGENT": "agent", "TOOL": "tool", "CHAIN": "chain",
|
|
18
|
+
"RETRIEVER": "retriever", "EMBEDDING": "embedding", "GUARDRAIL": "guardrail", "EVALUATOR": "evaluator"}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def trace_to_spans(tr):
|
|
22
|
+
"""Langfuse trace (with observations) -> canonical spans. The trace itself becomes the root span."""
|
|
23
|
+
tid = tr["id"]
|
|
24
|
+
md = tr.get("metadata") or {}
|
|
25
|
+
root = {
|
|
26
|
+
"trace_id": tid, "span_id": f"trace-{tid}", "parent_id": None, "name": tr.get("name") or "trace", "kind": "chain",
|
|
27
|
+
"start": parse_time(tr.get("timestamp")), "end": None, "status": "ok", "error": None,
|
|
28
|
+
"input": tr.get("input"), "output": tr.get("output"), "project": md.get("project") or tr.get("release"),
|
|
29
|
+
"environment": tr.get("environment") or md.get("environment"), "session_id": tr.get("sessionId"), "user_id": tr.get("userId"),
|
|
30
|
+
"tags": tr.get("tags") or [], "framework": "langfuse", "source": "langfuse",
|
|
31
|
+
"feedback": [{"key": s.get("name"), "score": s.get("value")} for s in tr.get("scores") or [] if isinstance(s.get("value"), (int, float))],
|
|
32
|
+
}
|
|
33
|
+
spans, ends = [root], []
|
|
34
|
+
for o in tr.get("observations") or []:
|
|
35
|
+
u = o.get("usageDetails") or o.get("usage") or {}
|
|
36
|
+
it = u.get("input") or u.get("promptTokens") or u.get("input_tokens") or 0
|
|
37
|
+
ot = u.get("output") or u.get("completionTokens") or u.get("output_tokens") or 0
|
|
38
|
+
cr = u.get("cache_read_input_tokens") or u.get("input_cached_tokens") or u.get("cache_read") or 0
|
|
39
|
+
cw = u.get("cache_creation_input_tokens") or 0
|
|
40
|
+
err = o.get("statusMessage") if o.get("level") == "ERROR" else None
|
|
41
|
+
end = parse_time(o.get("endTime"))
|
|
42
|
+
ends.append(end)
|
|
43
|
+
omd = o.get("metadata") or {}
|
|
44
|
+
out = o.get("output")
|
|
45
|
+
docs = None
|
|
46
|
+
if o.get("type") == "RETRIEVER" or "retriev" in (o.get("name") or "").lower():
|
|
47
|
+
d = out.get("documents") if isinstance(out, dict) else out
|
|
48
|
+
docs = len(d) if isinstance(d, list) else None
|
|
49
|
+
spans.append({
|
|
50
|
+
"trace_id": tid, "span_id": o["id"], "parent_id": o.get("parentObservationId") or root["span_id"],
|
|
51
|
+
"name": o.get("name") or o.get("type"), "kind": OBS_KIND.get(o.get("type"), "span"),
|
|
52
|
+
"start": parse_time(o.get("startTime")), "end": end, "status": "error" if err else "ok", "error": err,
|
|
53
|
+
"model": o.get("model"), "input_tokens": it, "output_tokens": ot, "cache_read": cr, "cache_write": cw,
|
|
54
|
+
"cost": o.get("calculatedTotalCost") if o.get("calculatedTotalCost") is not None else (o.get("costDetails") or {}).get("total"),
|
|
55
|
+
"ttft_ms": (parse_time(o.get("completionStartTime")) - parse_time(o.get("startTime"))) * 1000
|
|
56
|
+
if o.get("completionStartTime") and o.get("startTime") else None,
|
|
57
|
+
"stop_reason": omd.get("finish_reason"), "input": o.get("input"), "output": out,
|
|
58
|
+
"node": omd.get("langgraph_node"), "agent": omd.get("agent_name"), "framework": "langfuse", "source": "langfuse", "docs": docs,
|
|
59
|
+
})
|
|
60
|
+
valid = [e for e in ends if e]
|
|
61
|
+
root["end"] = max(valid) if valid else root["start"]
|
|
62
|
+
if any(s.get("status") == "error" for s in spans[1:]) and not tr.get("output"):
|
|
63
|
+
root["status"] = "error"
|
|
64
|
+
root["error"] = next(s["error"] for s in spans[1:] if s.get("status") == "error")
|
|
65
|
+
return spans
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class LangfusePuller:
|
|
69
|
+
def __init__(self, cfg, state):
|
|
70
|
+
self.host = (cfg.get("host") or "https://cloud.langfuse.com").rstrip("/")
|
|
71
|
+
pk = os.environ.get(cfg.get("public_key_env", "LANGFUSE_PUBLIC_KEY"), "")
|
|
72
|
+
sk = os.environ.get(cfg.get("secret_key_env", "LANGFUSE_SECRET_KEY"), "")
|
|
73
|
+
self.auth = "Basic " + base64.b64encode(f"{pk}:{sk}".encode()).decode()
|
|
74
|
+
self.state = state
|
|
75
|
+
self.lookback = float(cfg.get("lookback_hours", 24)) * 3600
|
|
76
|
+
|
|
77
|
+
def _get(self, path, params=None):
|
|
78
|
+
url = f"{self.host}{path}" + (("?" + urlencode(params)) if params else "")
|
|
79
|
+
req = urllib.request.Request(url, headers={"Authorization": self.auth, "Accept": "application/json"})
|
|
80
|
+
with urllib.request.urlopen(req, timeout=30) as r:
|
|
81
|
+
return json.loads(r.read())
|
|
82
|
+
|
|
83
|
+
def pull(self, max_traces=2000):
|
|
84
|
+
since = self.state.get("since") or (time.time() - self.lookback)
|
|
85
|
+
iso = datetime.fromtimestamp(since, timezone.utc).isoformat().replace("+00:00", "Z")
|
|
86
|
+
traces, page, newest = [], 1, since
|
|
87
|
+
while len(traces) < max_traces:
|
|
88
|
+
res = self._get("/api/public/traces", {"fromTimestamp": iso, "page": page, "limit": 50})
|
|
89
|
+
data = res.get("data") or []
|
|
90
|
+
for t in data:
|
|
91
|
+
traces.append(self._get(f"/api/public/traces/{t['id']}"))
|
|
92
|
+
newest = max(newest, parse_time(t.get("timestamp")) or 0)
|
|
93
|
+
meta = res.get("meta") or {}
|
|
94
|
+
if not data or page >= (meta.get("totalPages") or 1):
|
|
95
|
+
break
|
|
96
|
+
page += 1
|
|
97
|
+
self.state["since"] = max(since, newest - 600)
|
|
98
|
+
return traces
|